From 79af264358059ed14e692fb6b9e11efc3bd9b67c Mon Sep 17 00:00:00 2001 From: Mukul Sharma Date: Sun, 30 Aug 2026 07:16:41 +0530 Subject: [PATCH] removed files --- CLAUDE.md | 106 - claude/00-overview.md | 63 - claude/01-repo-structure.md | 70 - claude/02-cluster-fleet.md | 80 - claude/03-chart-inventory.md | 82 - claude/04-override-hierarchy.md | 72 - claude/05-deploy-lifecycle.md | 103 - claude/06-secrets-and-identity.md | 70 - claude/07-singletons-and-blast-radius.md | 85 - claude/08-pre-commit-and-hooks.md | 69 - claude/09-common-tasks.md | 67 - claude/10-glossary-and-references.md | 61 - contour-nodeselector-tolerations-summary.md | 70 - docs/architecture.md | 148 -- docs/global/AGENT_BOUNDARIES.md | 123 - docs/global/SANCTITY_RULES.md | 122 - docs/global/agent-operations-guide.md | 82 - docs/global/coding-guidelines/argocd.md | 67 - docs/global/coding-guidelines/helm-values.md | 167 -- .../global/coding-guidelines/observability.md | 78 - docs/global/escalation-matrix.md | 39 - docs/platform/procedures/add-contour-route.md | 231 -- .../procedures/blue-green-chart-migration.md | 206 -- docs/platform/procedures/deboard-app.md | 119 - .../procedures/fork-upstream-chart.md | 158 -- .../platform/procedures/modify-alert-rules.md | 227 -- .../procedures/modify-observability-config.md | 210 -- .../procedures/onboard-app-to-cluster.md | 204 -- .../procedures/onboard-new-cluster.md | 140 - .../procedures/update-chart-version.md | 176 -- docs/platform/runbooks/argocd-sync-failure.md | 231 -- docs/platform/runbooks/ingress-down.md | 197 -- docs/platform/runbooks/metrics-gap.md | 233 -- .../runbooks/pod-pending-scheduling.md | 192 -- docs/platform/runbooks/vault-unavailable.md | 251 -- docs/platform/schemas/custom-values-schema.md | 379 --- .../schemas/incubator-values-schema.md | 195 -- .../schemas/raw-manifest-sidecar-schema.md | 191 -- .../storageclass-priorityclass-schema.md | 158 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../victoria-metrics-agent/custom-values.yaml | 296 --- .../ingress-nginx-internal/custom-values.yaml | 26 - .../kube-state-metrics/custom-values.yaml | 478 ---- .../victoria-metrics-agent/custom-values.yaml | 296 --- .../custom-values.yaml | 234 -- .../custom-values.yaml | 291 --- .../custom-values.yaml | 316 --- .../kube-state-metrics/custom-values.yaml | 478 ---- .../victoria-metrics-agent/custom-values.yaml | 296 --- .../kube-state-metrics/custom-values.yaml | 478 ---- .../victoria-metrics-agent/custom-values.yaml | 296 --- .../kube-state-metrics/custom-values.yaml | 478 ---- .../victoria-metrics-agent/custom-values.yaml | 296 --- .../gke-central-prd-ase1a/README.md | 3 - .../ai-gateway-ext/custom-values.yaml | 295 --- .../ai-gateway/custom-values.yaml | 286 --- .../custom-values.yaml | 99 - .../aurva-dataplane/custom-values.yaml | 689 ----- .../cert-manager/custom-values.yaml | 129 - .../clickhouse/custom-values.yaml | 251 -- .../computeclass/azul-cc.yaml | 45 - .../computeclass/central-devops-cc.yaml | 45 - .../computeclass/central-kyverno-cc.yaml | 45 - .../computeclass/compactduo-cc.yaml | 72 - .../computeclass/compacttetra-cc.yaml | 45 - .../computeclass/contour-external-arm.yaml | 72 - .../computeclass/contour-external-cc.yaml | 72 - .../computeclass/contour-internal-0-arm.yaml | 72 - .../computeclass/contour-internal-0-cc.yaml | 45 - .../computeclass/contour-internal-1-arm.yaml | 72 - .../computeclass/contour-internal-1-cc.yaml | 72 - .../contour-internal-intra-0-arm.yaml | 72 - .../contour-internal-intra-0-cc.yaml | 72 - .../contour-internal-intra-1-arm.yaml | 72 - .../contour-internal-intra-1-cc.yaml | 45 - .../computeclass/contour-shared-arm.yaml | 72 - .../computeclass/contour-shared-cc.yaml | 45 - .../computeclass/devops-mcp-cc.yaml | 45 - .../computeclass/gatekeeper-cc.yaml | 45 - .../computeclass/loghouse-cc.yaml | 45 - .../computeclass/megaduo-cc.yaml | 72 - .../computeclass/megaduolite-cc.yaml | 72 - .../computeclass/megaoctalite-cc.yaml | 45 - .../computeclass/megatetra-cc.yaml | 72 - .../computeclass/megatetralite-cc.yaml | 72 - .../computeclass/megauno-cc.yaml | 45 - .../computeclass/megaunolite-cc.yaml | 45 - .../computeclass/mlp-g2-standard-8-cc.yaml | 45 - .../computeclass/sale-rescue-cc.yaml | 45 - .../computeclass/session-mgr-cc.yaml | 72 - .../computeclass/sumoduo-c4d-cc.yaml | 45 - .../computeclass/sumoduo-cc.yaml | 72 - .../computeclass/sumoduolite-cc.yaml | 45 - .../computeclass/sumotetra-cc.yaml | 45 - .../computeclass/sumouno-cc.yaml | 45 - .../computeclass/sumounolite-cc.yaml | 45 - .../computeclass/vmagent-cc.yaml | 45 - .../computeclass/vmagent-dr-cc.yaml | 45 - .../computeclass/vmagent-mds-cc.yaml | 72 - .../computeclass/vminsert-cc.yaml | 45 - .../computeclass/vminsert-mds-cc.yaml | 45 - .../computeclass/vmselect-cc.yaml | 45 - .../computeclass/vmselect-mds-cc.yaml | 72 - .../computeclass/vmstorage-cc.yaml | 45 - .../computeclass/vmstorage-mds-cc.yaml | 45 - .../computeclass/vmstorage-n4d-cc.yaml | 45 - .../computeclass/vmstorage-sale-24aug-cc.yaml | 45 - .../computeclass/vmstorage-sale-cc.yaml | 45 - .../conntrack-adjuster/custom-values.yaml | 25 - .../contour-ca-issuer/custom-values.yaml | 17 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-external-1/custom-values.yaml | 101 - .../contour-external/custom-values.yaml | 106 - .../contour-internal-0/custom-values.yaml | 109 - .../contour-internal-1/custom-values.yaml | 113 - .../custom-values.yaml | 106 - .../custom-values.yaml | 107 - .../coredns/custom-values.yaml | 28 - .../elasticsearch-mcp/custom-values.yaml | 95 - .../etcd/custom-values.yaml | 1105 -------- .../external-secrets/custom-values.yaml | 62 - .../fireworks-ai/custom-values.yaml | 279 -- .../flagger/custom-values.yaml | 68 - .../fluentd-copy/custom-values.yaml | 787 ------ .../fluentd/custom-values.yaml | 829 ------ .../keda/custom-values.yaml | 42 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../kubectl-mcp-server/custom-values.yaml | 144 -- .../kyverno/custom-values.yaml | 2244 ----------------- .../kyverno/policies/protect-namespace.yaml | 40 - .../kyverno/policies/restrict-replicas.yaml | 29 - .../loadtester/custom-values.yaml | 116 - .../custom-values.yaml | 338 --- .../custom-values.yaml | 330 --- .../custom-values.yaml | 277 -- .../custom-values.yaml | 119 - .../custom-values.yaml | 497 ---- .../custom-values.yaml | 170 -- .../telegraf-operator/custom-values.yaml | 150 -- .../temporal/custom-values.yaml | 157 -- .../custom-values.yaml | 296 --- .../custom-values.yaml | 272 -- .../custom-values.yaml | 304 --- .../custom-values.yaml | 340 --- .../custom-values.yaml | 340 --- .../victoriametrics-agent/custom-values.yaml | 484 ---- .../victoriametrics-insert/custom-values.yaml | 411 --- .../victoriametrics-select/custom-values.yaml | 449 ---- .../custom-values.yaml | 363 --- .../gke-dataengg-prd-ase1a/README.md | 3 - .../aurva-dataplane/custom-values.yaml | 684 ----- .../cert-manager/custom-values.yaml | 129 - .../16c-104g-dp-dpcon-tco-farm-cc.yaml | 22 - .../16c-104g-farm-dp-dpcon-tworker-cc.yaml | 22 - .../52c-150g-deng-dpexp-ab-cc.yaml | 22 - .../8c-64g-dp-dpnrt-druid-cc.yaml | 28 - .../8c-64g-dp-dpnrt-druid-od-cc.yaml | 22 - .../computeclass/alluxio-32c-cc.yaml | 22 - .../alluxio-hmem32-8-lssd-prd-cc.yaml | 23 - .../c3d-hmem-8-sp-dpnrt-druid-int-cc.yaml | 22 - .../computeclass/compactocta-cc.yaml | 22 - .../computeclass/compacttetra-cc.yaml | 28 - .../computeclass/contour-external-cc-v1.yaml | 28 - .../computeclass/contour-external-cc.yaml | 28 - .../contour-internal-0-cc-v1.yaml | 28 - .../computeclass/contour-internal-0-cc.yaml | 28 - .../contour-internal-1-cc-v1.yaml | 28 - .../computeclass/contour-internal-1-cc.yaml | 28 - .../computeclass/contour-intra-0-cc-v1.yaml | 28 - .../computeclass/contour-intra-1-cc-v1.yaml | 28 - .../computeclass/contour-intra-1-cc.yaml | 28 - .../computeclass/contour-shared-cc-v1.yaml | 28 - .../computeclass/contour-shared-cc.yaml | 22 - .../computeclass/dataengg-devops.yaml | 34 - .../computeclass/dp-airflow-cc.yaml | 22 - .../computeclass/dp-dpnrt-zookeeper-cc.yaml | 22 - .../dpcon-alluxio-n2-hmem-8-c-od.yaml | 25 - .../dpcon-n2-hmem-32-a-co-cc.yaml | 28 - .../dpcon-n2-hmem-32-b-co-cc.yaml | 28 - .../dpcon-n2-hmem-32-c-co-cc.yaml | 28 - .../computeclass/dpcon-n2-hmem-48-a-cc.yaml | 43 - .../dpcon-n2-hmem-48-a-co-cc.yaml | 28 - .../computeclass/dpcon-n2-hmem-48-b-cc.yaml | 40 - .../dpcon-n2-hmem-48-b-co-cc.yaml | 28 - .../computeclass/dpcon-n2-hmem-48-c-cc.yaml | 40 - .../dpcon-n2-hmem-48-c-co-cc.yaml | 28 - .../computeclass/dpcon-n2-hmem-48-c-w-cc.yaml | 28 - .../computeclass/dpcon-n2d-hmem-16-b-cc.yaml | 22 - .../dpcon-n2d-hmem-32-b-od-cc.yaml | 22 - .../dpcon-n2d-hmem-48-a-sensitive-cc.yaml | 22 - .../computeclass/dpcon-n2d-hmem-48-b-cc.yaml | 22 - .../computeclass/dpcon-n2d-hmem-48-c-cc.yaml | 22 - .../computeclass/kuberay-operator-cc.yaml | 22 - .../computeclass/megaduo-cc.yaml | 22 - .../computeclass/megaduolite-cc.yaml | 22 - .../computeclass/megaquad-cc.yaml | 22 - .../computeclass/megatetra-cc.yaml | 22 - .../n2-hmem-64-ondemand-a-cc.yaml | 33 - .../n2d-highmem-8-dp-dpcon-zep-cc.yaml | 22 - .../computeclass/n2d-hmem-16-od-ls-cc.yaml | 22 - .../computeclass/n2d-hmem-8-sp-ls-cc.yaml | 22 - .../computeclass/n4-hmem-48-spot-cc.yaml | 27 - .../computeclass/nginx-shared-cc.yaml | 28 - .../np-dp-kuberay-4c-16g-prd-ase1-cc.yaml | 22 - .../os-trino-poc-n2d-dp-dpcon-wk-cc.yaml | 22 - .../computeclass/strimzi-kafka-cc.yaml | 22 - .../computeclass/sumoduo-cc.yaml | 28 - .../computeclass/sumoduolite-cc.yaml | 22 - .../computeclass/sumounolite-cc.yaml | 28 - .../computeclass/vmagent-mds-cc.yaml | 22 - .../computeclass/vminsert-mds-cc.yaml | 22 - .../computeclass/vmselect-mds-cc.yaml | 22 - .../computeclass/vmstack-startree-cc.yaml | 22 - .../computeclass/vmstorage-n4d-cc.yaml | 22 - .../computeclass/warpstream-dp-cc.yaml | 23 - .../wk-trino-spot-48c-384g-prd-ase1-cc.yaml | 28 - .../conntrack-adjuster/custom-values.yaml | 17 - .../contour-ca-issuer/custom-values.yaml | 17 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-external/custom-values.yaml | 114 - .../contour-internal-0/custom-values.yaml | 117 - .../contour-internal-1/custom-values.yaml | 117 - .../custom-values.yaml | 114 - .../custom-values.yaml | 114 - .../coredns/custom-values.yaml | 28 - .../external-secrets/custom-values.yaml | 61 - .../flagger/custom-values.yaml | 62 - .../fluentd/custom-values.yaml | 730 ------ .../ingress-nginx-external/custom-values.yaml | 34 - .../ingress-nginx-internal/custom-values.yaml | 33 - .../ingress-nginx-secured/custom-values.yaml | 33 - .../keda/custom-values.yaml | 26 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 427 ---- .../kubectl-mcp-server/custom-values.yaml | 119 - .../kyverno/custom-values.yaml | 2243 ---------------- .../kyverno/policies/protect-namespace.yaml | 40 - .../kyverno/policies/restrict-replicas.yaml | 29 - .../loadtester/custom-values.yaml | 116 - .../custom-values.yaml | 281 --- .../custom-values.yaml | 134 - .../custom-values.yaml | 487 ---- .../custom-values.yaml | 169 -- .../telegraf-operator/custom-values.yaml | 151 -- .../custom-values.yaml | 296 --- .../custom-values.yaml | 272 -- .../custom-values.yaml | 304 --- .../custom-values.yaml | 340 --- .../custom-values.yaml | 340 --- .../custom-values.yaml | 234 -- .../custom-values.yaml | 293 --- .../custom-values.yaml | 316 --- .../victoriametrics-agent/custom-values.yaml | 488 ---- .../victoriametrics-insert/custom-values.yaml | 411 --- .../victoriametrics-select/custom-values.yaml | 447 ---- .../custom-values.yaml | 363 --- .../gke-datascience-prd-as1a/README.md | 3 - .../alloy/custom-values.yaml | 50 - .../aurva-dataplane/custom-values.yaml | 674 ----- .../cert-manager/custom-values.yaml | 129 - .../c3-highcpu-44-compute-class.yaml | 23 - .../c3d-highcpu-30-compute-class.yaml | 23 - .../c4a-highcpu-16-compute-class.yaml | 23 - .../computeclass/contour-internal-0-arm.yaml | 28 - .../computeclass/contour-internal-1-arm.yaml | 28 - .../contour-internal-dataproc-arm.yaml | 22 - .../contour-internal-intra-0-arm.yaml | 28 - .../contour-internal-intra-1-arm.yaml | 28 - .../computeclass/contour-shared-arm.yaml | 28 - .../computeclass/datascience-devops.yaml | 34 - .../g2-standard-16-l4-compute-class.yaml | 49 - .../g2-standard-4-l4-compute-class.yaml | 49 - .../g2-standard-8-l4-300gb-compute-class.yaml | 52 - .../g2-standard-8-l4-compute-class.yaml | 49 - .../n2d-standard-48-4lssd-compute-class.yaml | 25 - .../conntrack-adjuster/custom-values.yaml | 20 - .../contour-ca-issuer/custom-values.yaml | 17 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-internal-0/custom-values.yaml | 118 - .../contour-internal-1/custom-values.yaml | 118 - .../custom-values.yaml | 112 - .../custom-values.yaml | 114 - .../custom-values.yaml | 115 - .../coredns/custom-values.yaml | 28 - .../coroot-node-agent/custom-values.yaml | 52 - .../external-secrets/custom-values.yaml | 61 - .../flagger/custom-values.yaml | 62 - .../fluentd/custom-values.yaml | 731 ------ .../ingress-nginx-internal/custom-values.yaml | 26 - .../keda/custom-values.yaml | 46 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../kubectl-mcp-server/custom-values.yaml | 117 - .../kubernetes-dashboard/custom-values.yaml | 442 ---- .../kyverno/custom-values.yaml | 2240 ---------------- .../kyverno/policies/protect-namespace.yaml | 40 - .../kyverno/policies/restrict-replicas.yaml | 29 - .../loadtester/custom-values.yaml | 116 - .../custom-values.yaml | 281 --- .../custom-values.yaml | 134 - .../paused-container/custom-values.yaml | 82 - .../custom-values.yaml | 496 ---- .../custom-values.yaml | 170 -- .../telegraf-operator/custom-values.yaml | 150 -- .../custom-values.yaml | 295 --- .../custom-values.yaml | 272 -- .../victoria-metrics-agent/custom-values.yaml | 295 --- .../custom-values.yaml | 304 --- .../custom-values.yaml | 340 --- .../custom-values.yaml | 305 --- .../custom-values.yaml | 340 --- .../victoria-metrics-alert/custom-values.yaml | 305 --- .../custom-values.yaml | 234 -- .../custom-values.yaml | 293 --- .../custom-values.yaml | 316 --- .../victoriametrics-agent/custom-values.yaml | 485 ---- .../victoriametrics-insert/custom-values.yaml | 411 --- .../victoriametrics-select/custom-values.yaml | 445 ---- .../custom-values.yaml | 363 --- .../gke-datascience-prd-ase1a/README.md | 3 - .../alloy/custom-values.yaml | 53 - .../aurva-dataplane/custom-values.yaml | 694 ----- .../cert-manager/custom-values.yaml | 129 - .../computeclass/c3-standard-22-lssd-cc.yaml | 23 - .../computeclass/contour-internal-0-cc.yaml | 28 - .../computeclass/contour-internal-1-cc.yaml | 34 - .../contour-internal-dataproc-cc.yaml | 28 - .../computeclass/contour-intra-0-cc.yaml | 28 - .../computeclass/contour-intra-1-cc.yaml | 28 - .../computeclass/contour-shared-cc.yaml | 28 - .../computeclass/datascience-devops-cc.yaml | 22 - .../computeclass/datascience-kyverno-cc.yaml | 22 - .../computeclass/ds-airflow-cc.yaml | 22 - .../computeclass/g2-l4-priority-class-cc.yaml | 26 - .../g2-standard-4-l4-compute-class-cc.yaml | 26 - ...-standard-8-l4-300gb-compute-class-cc.yaml | 26 - .../g2-standard-8-l4-compute-class-cc.yaml | 54 - .../computeclass/load-testing-v2-mlp-cc.yaml | 22 - .../computeclass/megaduo-cc.yaml | 34 - .../computeclass/megaduo-ctz-cc.yaml | 22 - .../computeclass/megaduo-rto-consumer-cc.yaml | 22 - .../computeclass/megaduolite-cc.yaml | 22 - .../computeclass/megaquad-cc.yaml | 22 - .../computeclass/megatetra-cc.yaml | 22 - .../computeclass/megatetralite-cc.yaml | 22 - .../computeclass/mlp-a2-highgpu-1-cc.yaml | 26 - .../computeclass/mlp-c3d-hc-30-300-v1-cc.yaml | 28 - .../mlp-c3d-highcpu-30-sz-cc.yaml | 22 - .../mlp-c3d-highcpu-30-v1-cc.yaml | 22 - .../computeclass/mlp-c4d-localssd-cc.yaml | 22 - .../mlp-g2-standard-16-v2-cc.yaml | 26 - .../mlp-g2-standard-32-custom-cc.yaml | 26 - .../mlp-g2-standard-8-zone-a-cc.yaml | 26 - .../n2d-standard-48-4lssd-cc.yaml | 23 - .../computeclass/nginx-internal-cc.yaml | 22 - .../np-dsci-ml-g2-standard-8-prd-ase1-cc.yaml | 26 - .../computeclass/rockdb-localssd-1-cc.yaml | 22 - .../computeclass/rust-onboard-cc.yaml | 22 - .../computeclass/ssd-cosmos-v2-cc.yaml | 22 - .../computeclass/sumoduo-cc.yaml | 34 - .../computeclass/sumotetra-cc.yaml | 22 - .../computeclass/vmagent-dr-cc.yaml | 22 - .../computeclass/vmagent-mds-cc.yaml | 22 - .../computeclass/vmagent-n4-cc.yaml | 22 - .../computeclass/vmagent-n4d-cc.yaml | 34 - .../computeclass/vminsert-cc.yaml | 22 - .../computeclass/vminsert-mds-cc.yaml | 22 - .../computeclass/vmselect-cc.yaml | 22 - .../computeclass/vmselect-mds-cc.yaml | 22 - .../computeclass/vmstorage-c4d-cc.yaml | 22 - .../computeclass/vmstorage-n4-cc.yaml | 22 - .../computeclass/vmstorage-n4d-cc.yaml | 28 - .../computeclass/vmstorage-sale-24aug-cc.yaml | 22 - .../conntrack-adjuster/custom-values.yaml | 20 - .../contour-ca-issuer/custom-values.yaml | 17 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-internal-0/custom-values.yaml | 127 - .../contour-internal-1/custom-values.yaml | 118 - .../custom-values.yaml | 112 - .../custom-values.yaml | 125 - .../custom-values.yaml | 116 - .../coredns/custom-values.yaml | 28 - .../coroot-node-agent/custom-values.yaml | 54 - .../external-secrets/custom-values.yaml | 61 - .../flagger/custom-values.yaml | 63 - .../fluentd/custom-values.yaml | 740 ------ .../ingress-nginx-internal/custom-values.yaml | 26 - .../keda/custom-values.yaml | 45 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../kubectl-mcp-server/custom-values.yaml | 117 - .../kubernetes-dashboard/custom-values.yaml | 442 ---- .../kyverno/custom-values.yaml | 2244 ----------------- .../kyverno/policies/protect-namespace.yaml | 40 - .../kyverno/policies/restrict-replicas.yaml | 29 - .../loadtester/custom-values.yaml | 115 - .../custom-values.yaml | 153 -- .../custom-values.yaml | 496 ---- .../custom-values.yaml | 170 -- .../telegraf-operator/custom-values.yaml | 232 -- .../custom-values.yaml | 295 --- .../custom-values.yaml | 272 -- .../custom-values.yaml | 304 --- .../custom-values.yaml | 332 --- .../custom-values.yaml | 340 --- .../custom-values.yaml | 488 ---- .../custom-values.yaml | 411 --- .../custom-values.yaml | 449 ---- .../custom-values.yaml | 363 --- helm-overrides/gke-demand-prd-ase1a/README.md | 3 - .../alloy/custom-values.yaml | 50 - .../aurva-dataplane/custom-values.yaml | 686 ----- .../cert-manager/custom-values.yaml | 129 - .../computeclass/azul-cc.yaml | 24 - .../computeclass/compactduo-cc.yaml | 45 - .../computeclass/compacttetra-cc.yaml | 45 - .../computeclass/contour-external-cc-v1.yaml | 30 - .../computeclass/contour-external-cc.yaml | 24 - .../contour-internal-0-cc-v1.yaml | 30 - .../computeclass/contour-internal-0-cc.yaml | 30 - .../contour-internal-1-cc-v1.yaml | 30 - .../computeclass/contour-internal-1-cc.yaml | 30 - .../computeclass/contour-intra-0-cc-v1.yaml | 30 - .../computeclass/contour-intra-0-cc.yaml | 30 - .../computeclass/contour-intra-1-cc-v1.yaml | 24 - .../computeclass/contour-intra-1-cc.yaml | 30 - .../computeclass/contour-shared-cc-v1.yaml | 30 - .../computeclass/contour-shared-cc.yaml | 24 - .../computeclass/demand-devops-cc.yaml | 45 - .../computeclass/demand-kyverno-cc.yaml | 45 - .../computeclass/dmnd-c4a-32c64g-cc.yaml | 45 - .../computeclass/dns-coldstart-probe-cc.yaml | 51 - .../computeclass/megaduo-cc.yaml | 72 - .../computeclass/megaduo-spp-cc.yaml | 72 - .../computeclass/megaduolite-cc.yaml | 72 - .../computeclass/megatetra-cc.yaml | 45 - .../computeclass/megatetra-trnst-cc.yaml | 45 - .../computeclass/megatetralite-cc.yaml | 45 - .../computeclass/pdp-relay-v1-cc.yaml | 24 - .../prod-comms-consumer-notification-cc.yaml | 24 - .../prod-comms-consumer-v1-cc.yaml | 24 - .../computeclass/search-relay-cc.yaml | 24 - .../computeclass/sumoduo-azul-cc.yaml | 45 - .../computeclass/sumoduo-cc.yaml | 72 - .../computeclass/sumoduolite-azul-cc.yaml | 45 - .../computeclass/sumoduolite-cc.yaml | 45 - .../computeclass/sumoocta-cc.yaml | 45 - .../computeclass/sumotetra-cc.yaml | 45 - .../computeclass/sumotetralite-cc.yaml | 72 - .../computeclass/vm-agent-cc.yaml | 24 - .../computeclass/vmagent-c4d-cc.yaml | 24 - .../computeclass/vminsert-mds-cc.yaml | 24 - .../computeclass/vmselect-mds-cc.yaml | 24 - .../computeclass/vmstorage-n4d-cc.yaml | 24 - .../conntrack-adjuster/custom-values.yaml | 18 - .../contour-ca-issuer/custom-values.yaml | 17 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-external/custom-values.yaml | 111 - .../contour-internal-0/custom-values.yaml | 120 - .../contour-internal-1/custom-values.yaml | 117 - .../custom-values.yaml | 118 - .../custom-values.yaml | 115 - .../coredns/custom-values.yaml | 28 - .../coroot-node-agent/custom-values.yaml | 52 - .../external-secrets/custom-values.yaml | 61 - .../flagger/custom-values.yaml | 62 - .../fluentd/custom-values.yaml | 734 ------ .../keda/custom-values.yaml | 61 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 429 ---- .../kubectl-mcp-server/custom-values.yaml | 119 - .../kyverno/custom-values.yaml | 2244 ----------------- .../kyverno/policies/protect-namespace.yaml | 40 - .../kyverno/policies/restrict-replicas.yaml | 29 - .../loadtester/custom-values.yaml | 116 - .../node-thp-config/custom-values.yaml | 9 - .../custom-values.yaml | 281 --- .../custom-values.yaml | 133 - .../custom-values.yaml | 487 ---- .../custom-values.yaml | 169 -- .../telegraf-operator/custom-values.yaml | 292 --- .../custom-values.yaml | 296 --- .../custom-values.yaml | 272 -- .../custom-values.yaml | 304 --- .../custom-values.yaml | 340 --- .../custom-values.yaml | 340 --- .../victoriametrics-agent/custom-values.yaml | 488 ---- .../victoriametrics-insert/custom-values.yaml | 411 --- .../victoriametrics-select/custom-values.yaml | 449 ---- .../custom-values.yaml | 363 --- helm-overrides/gke-dsgpu-prd-ase1a/README.md | 3 - .../aurva-dataplane/custom-values.yaml | 661 ----- .../cert-manager/custom-values.yaml | 129 - .../a2-highgpu-1-imgcache-compute-class.yaml | 52 - ...u-22-contour-internal-0-compute-class.yaml | 24 - ...u-22-contour-internal-1-compute-class.yaml | 24 - ...ighcpu-4-nginx-internal-compute-class.yaml | 24 - .../c3-highcpu-44-vmselect-compute-class.yaml | 24 - ...pu-8-contour-internal-0-compute-class.yaml | 24 - ...pu-8-contour-internal-1-compute-class.yaml | 24 - ...u-16-contour-internal-0-compute-class.yaml | 24 - ...u-16-contour-internal-1-compute-class.yaml | 24 - .../computeclass/datascience-devops.yaml | 36 - .../computeclass/dsgpu-kyverno-cc.yaml | 24 - ...rd-4-datascience-devops-compute-class.yaml | 24 - .../computeclass/g2-l4-priority-class.yaml | 51 - .../g2-standard-16-l4-compute-class-ld.yaml | 51 - .../g2-standard-16-l4-compute-class.yaml | 51 - ...dard-4-l4-compute-class-driver-latest.yaml | 51 - .../g2-standard-4-l4-compute-class.yaml | 51 - .../g2-standard-8-l4-300gb-compute-class.yaml | 54 - .../g2-standard-8-l4-compute-class.yaml | 51 - .../g2-std-16-l4-priority-class.yaml | 51 - .../n4-highcpu-16-vminsert-compute-class.yaml | 24 - .../n4d-highcpu-64-vmagent-compute-class.yaml | 24 - ...4d-highmem-48-vmstorage-compute-class.yaml | 24 - .../conntrack-adjuster/custom-values.yaml | 26 - .../contour-ca-issuer/custom-values.yaml | 17 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-internal-0/custom-values.yaml | 103 - .../contour-internal-1/custom-values.yaml | 118 - .../custom-values.yaml | 101 - .../custom-values.yaml | 116 - .../coredns/custom-values.yaml | 27 - ...u-prd-envoy-headless-external-dns-svc.yaml | 30 - .../external-dns/custom-values.yaml | 60 - .../external-secrets/custom-values.yaml | 61 - .../flagger/custom-values.yaml | 62 - .../fluentd/custom-values.yaml | 731 ------ .../ingress-nginx-internal/custom-values.yaml | 26 - .../keda/custom-values.yaml | 46 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 148 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../kubectl-mcp-server/custom-values.yaml | 121 - .../kyverno/custom-values.yaml | 2244 ----------------- .../kyverno/policies/protect-namespace.yaml | 40 - .../kyverno/policies/restrict-replicas.yaml | 29 - .../loadtester/custom-values.yaml | 116 - .../custom-values.yaml | 281 --- .../custom-values.yaml | 134 - .../paused-container/custom-values.yaml | 82 - .../custom-values.yaml | 496 ---- .../custom-values.yaml | 170 -- .../telegraf-operator/custom-values.yaml | 150 -- .../custom-values.yaml | 272 -- .../custom-values.yaml | 340 --- .../victoriametrics-agent/custom-values.yaml | 480 ---- .../victoriametrics-insert/custom-values.yaml | 411 --- .../victoriametrics-select/custom-values.yaml | 444 ---- .../custom-values.yaml | 363 --- .../gke-farmiso-prd-ase1a/README.md | 3 - .../aurva-dataplane/custom-values.yaml | 677 ----- .../cert-manager/custom-values.yaml | 129 - .../computeclass/compacttetra-cc.yaml | 29 - .../computeclass/contour-external-cc-v1.yaml | 28 - .../contour-internal-0-cc-v1.yaml | 28 - .../computeclass/contour-intra-0-cc-v1.yaml | 28 - .../computeclass/contour-shared-cc-v1.yaml | 28 - .../computeclass/contour-shared-cc.yaml | 28 - .../computeclass/farmiso-devops.yaml | 34 - .../computeclass/megaquad-cc.yaml | 29 - .../computeclass/vmagent-mds-cc.yaml | 22 - .../computeclass/vminsert-mds-cc.yaml | 22 - .../computeclass/vmselect-mds-cc.yaml | 22 - .../computeclass/vmstorage-n4d-cc.yaml | 22 - .../conntrack-adjuster/custom-values.yaml | 17 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-external/custom-values.yaml | 115 - .../contour-internal-0/custom-values.yaml | 118 - .../custom-values.yaml | 115 - .../coredns/custom-values.yaml | 28 - .../external-secrets/custom-values.yaml | 61 - .../flagger/custom-values.yaml | 62 - .../fluentd/custom-values.yaml | 812 ------ .../keda/custom-values.yaml | 26 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 427 ---- .../kubectl-mcp-server/custom-values.yaml | 119 - .../kyverno/custom-values.yaml | 2243 ---------------- .../kyverno/policies/protect-namespace.yaml | 40 - .../kyverno/policies/restrict-replicas.yaml | 29 - .../loadtester/custom-values.yaml | 116 - .../custom-values.yaml | 281 --- .../custom-values.yaml | 134 - .../custom-values.yaml | 487 ---- .../custom-values.yaml | 169 -- .../telegraf-operator/custom-values.yaml | 151 -- .../custom-values.yaml | 296 --- .../custom-values.yaml | 272 -- .../custom-values.yaml | 304 --- .../custom-values.yaml | 340 --- .../victoriametrics-agent/custom-values.yaml | 482 ---- .../victoriametrics-insert/custom-values.yaml | 411 --- .../victoriametrics-select/custom-values.yaml | 448 ---- .../custom-values.yaml | 363 --- helm-overrides/gke-supply-prd-ase1a/README.md | 3 - .../alloy/custom-values.yaml | 53 - .../aurva-dataplane/custom-values.yaml | 694 ----- .../cert-manager/custom-values.yaml | 129 - .../computeclass/compactduo-cc.yaml | 24 - .../computeclass/compactocta-cc.yaml | 24 - .../computeclass/compacttetra-cc.yaml | 24 - .../computeclass/contour-external-cc.yaml | 30 - .../computeclass/contour-internal-0-cc.yaml | 30 - .../computeclass/contour-internal-1-cc.yaml | 30 - .../contour-internal-1-new-cc.yaml | 30 - .../computeclass/contour-intra-0-cc.yaml | 30 - .../computeclass/contour-intra-1-cc.yaml | 24 - .../computeclass/contour-shared-cc.yaml | 30 - .../computeclass/deepgram-api-pool-cc.yaml | 24 - .../computeclass/deepgram-proxy-pool-cc.yaml | 24 - .../computeclass/dg-eg-pool-cc.yaml | 28 - .../computeclass/efficient-ai-cc.yaml | 24 - .../computeclass/exp-cx-gen-cc.yaml | 30 - .../computeclass/g2-standard-4-cc.yaml | 24 - .../computeclass/megaduo-cc.yaml | 24 - .../computeclass/megaduolite-cc.yaml | 24 - .../computeclass/megaoctalite-cc.yaml | 30 - .../computeclass/megatetra-cc.yaml | 24 - .../computeclass/megatetralite-azul-cc.yaml | 24 - .../computeclass/megatetralite-cc.yaml | 30 - .../computeclass/megauno-cc.yaml | 24 - .../computeclass/megaunolite-cc.yaml | 24 - .../np-trino-n2-hmem-32-a-co-cc.yaml | 24 - .../np-trino-n2-hmem-32-a-sp-cc.yaml | 24 - .../computeclass/sumoduo-cc.yaml | 30 - .../computeclass/sumoduo-cpu-mngr-cc.yaml | 66 - .../computeclass/sumoduolite-cc.yaml | 30 - .../computeclass/sumoduolite-op-cc.yaml | 30 - .../computeclass/sumoocta-cc.yaml | 24 - .../computeclass/sumotetra-cc.yaml | 42 - .../computeclass/sumotetra-taxonomy-cc.yaml | 30 - .../computeclass/sumotetra-trnst-cc.yaml | 24 - .../computeclass/sumotetralite-cc.yaml | 30 - .../computeclass/sumouno-cc.yaml | 24 - .../computeclass/sumounolite-azul-cc.yaml | 30 - .../computeclass/sumounolite-cc.yaml | 30 - .../computeclass/supply-devops-cc.yaml | 24 - .../computeclass/supply-kyverno-cc.yaml | 24 - .../computeclass/vm-stack-ht-cc.yaml | 24 - .../computeclass/vmagent-c4-cc.yaml | 24 - .../computeclass/vmagent-cc.yaml | 24 - .../computeclass/vmagent-mds-cc.yaml | 42 - .../computeclass/vmagent-n4-cc.yaml | 24 - .../computeclass/vminsert-cc.yaml | 24 - .../computeclass/vminsert-mds-cc.yaml | 30 - .../computeclass/vmselect-cc.yaml | 24 - .../computeclass/vmselect-mds-cc.yaml | 30 - .../computeclass/vmstorage-c4d-cc.yaml | 25 - .../computeclass/vmstorage-cc.yaml | 24 - .../computeclass/vmstorage-mds-cc.yaml | 24 - .../computeclass/vmstorage-n4-cc.yaml | 24 - .../computeclass/vmstorage-n4d-cc.yaml | 30 - .../computeclass/vmstorage-sale-24aug-cc.yaml | 24 - .../computeclass/vmstorage-tmp-cc.yaml | 24 - .../conntrack-adjuster/custom-values.yaml | 20 - .../contour-ca-issuer/custom-values.yaml | 17 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-external/custom-values.yaml | 112 - .../contour-internal-0/custom-values.yaml | 125 - .../contour-internal-1/custom-values.yaml | 118 - .../custom-values.yaml | 123 - .../custom-values.yaml | 116 - .../coredns/custom-values.yaml | 28 - .../coroot-node-agent/custom-values.yaml | 54 - .../deepgram-onprem-v2/custom-values.yaml | 410 --- .../deepgram-onprem/custom-values.yaml | 918 ------- .../external-secrets/custom-values.yaml | 61 - .../flagger/custom-values.yaml | 63 - .../fluentd/custom-values.yaml | 854 ------- .../keda/custom-values.yaml | 45 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../kubectl-mcp-server/custom-values.yaml | 117 - .../kubernetes-dashboard/custom-values.yaml | 442 ---- .../kyverno/custom-values.yaml | 2244 ----------------- .../kyverno/policies/protect-namespace.yaml | 40 - .../kyverno/policies/restrict-replicas.yaml | 29 - .../loadtester/custom-values.yaml | 116 - .../node-thp-config/custom-values.yaml | 9 - .../custom-values.yaml | 288 --- .../custom-values.yaml | 110 - .../custom-values.yaml | 138 - .../custom-values.yaml | 496 ---- .../custom-values.yaml | 170 -- .../custom-values.yaml | 150 -- .../telegraf-operator/custom-values.yaml | 232 -- .../custom-values.yaml | 295 --- .../custom-values.yaml | 272 -- .../custom-values.yaml | 304 --- .../custom-values.yaml | 332 --- .../custom-values.yaml | 340 --- .../victoriametrics-agent/custom-values.yaml | 488 ---- .../victoriametrics-insert/custom-values.yaml | 411 --- .../victoriametrics-select/custom-values.yaml | 449 ---- .../custom-values.yaml | 363 --- .../argocd-admin-prd/custom-values.yaml | 274 +- .../contour-internal/custom-values.yaml | 77 - .../rancher/custom-values.yaml | 17 - .../k8s-central-mqkafka-prd-ase1/README.md | 3 - .../backend-nginx-cluster/custom-values.yaml | 241 -- .../cert-manager/custom-values.yaml | 129 - .../conntrack-adjuster/custom-values.yaml | 26 - .../contour-internal-0/custom-values.yaml | 97 - .../custom-values.yaml | 75 - .../external-secrets/custom-values.yaml | 50 - .../flagger/custom-values.yaml | 55 - .../fluentd/custom-values.yaml | 720 ------ .../custom-values.yaml | 43 - .../ingress-nginx-internal/custom-values.yaml | 43 - .../keda/custom-values.yaml | 38 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../nginx-demand/custom-values.yaml | 228 -- .../nginx-supply/custom-values.yaml | 206 -- .../custom-values.yaml | 496 ---- .../victoria-metrics-agent/custom-values.yaml | 295 --- .../victoria-metrics-alert/custom-values.yaml | 304 --- .../custom-values.yaml | 234 -- .../custom-values.yaml | 295 --- .../custom-values.yaml | 316 --- .../victoriametrics-agent/custom-values.yaml | 372 --- .../victoriametrics-insert/custom-values.yaml | 352 --- .../victoriametrics-select/custom-values.yaml | 409 --- .../custom-values.yaml | 393 --- .../contour-cert-checker/custom-values.yaml | 28 - .../contour-external/custom-values.yaml | 85 - .../contour-internal-0/custom-values.yaml | 89 - .../contour-internal-1/custom-values.yaml | 90 - .../coredns/custom-values.yaml | 27 - .../external-secrets/custom-values.yaml | 50 - .../flagger/custom-values.yaml | 68 - .../fluentd/custom-values.yaml | 719 ------ .../keda/custom-values.yaml | 35 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../custom-values.yaml | 119 - .../custom-values.yaml | 498 ---- .../custom-values.yaml | 170 -- .../telegraf-operator/custom-values.yaml | 151 -- .../victoria-metrics-agent/custom-values.yaml | 297 --- .../victoria-metrics-alert/custom-values.yaml | 305 --- .../custom-values.yaml | 235 -- .../custom-values.yaml | 294 --- .../custom-values.yaml | 317 --- helm-overrides/k8s-central-prd-ase1/README.md | 3 - .../ai-gateway/custom-values.yaml | 287 --- .../custom-values.yaml | 99 - .../alloy/custom-values.yaml | 53 - .../argocd/custom-values.yaml | 130 - .../aurva-dataplane/custom-values.yaml | 675 ----- .../cert-manager/custom-values.yaml | 129 - .../clickhouse/custom-values.yaml | 251 -- .../computeclass/contour-external-arm.yaml | 28 - .../computeclass/contour-external-cc.yaml | 34 - .../computeclass/contour-internal-0-arm.yaml | 28 - .../computeclass/contour-internal-0-cc.yaml | 22 - .../computeclass/contour-internal-1-arm.yaml | 28 - .../computeclass/contour-internal-1-cc.yaml | 34 - .../contour-internal-intra-0-arm.yaml | 28 - .../contour-internal-intra-0-cc.yaml | 22 - .../contour-internal-intra-1-arm.yaml | 28 - .../contour-internal-intra-1-cc.yaml | 22 - .../computeclass/contour-shared-arm.yaml | 28 - .../computeclass/contour-shared-cc.yaml | 22 - .../conntrack-adjuster/custom-values.yaml | 25 - .../contour-ca-issuer/custom-values.yaml | 17 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-external-1/custom-values.yaml | 101 - .../contour-external/custom-values.yaml | 101 - .../contour-internal-0/custom-values.yaml | 104 - .../contour-internal-1/custom-values.yaml | 108 - .../custom-values.yaml | 101 - .../custom-values.yaml | 102 - .../coredns/custom-values.yaml | 28 - .../coroot-node-agent/custom-values.yaml | 38 - .../elasticsearch-mcp/custom-values.yaml | 95 - .../etcd/custom-values.yaml | 1105 -------- .../external-secrets/custom-values.yaml | 62 - .../fireworks-ai/custom-values.yaml | 278 -- .../flagger/custom-values.yaml | 68 - .../fluentd-copy/custom-values.yaml | 787 ------ .../fluentd/custom-values.yaml | 829 ------ .../keda/custom-values.yaml | 42 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../kubectl-mcp-server/custom-values.yaml | 144 -- .../kyverno/custom-values.yaml | 2240 ---------------- .../kyverno/policies/protect-namespace.yaml | 40 - .../kyverno/policies/restrict-replicas.yaml | 29 - .../loadtester/custom-values.yaml | 116 - .../custom-values.yaml | 338 --- .../custom-values.yaml | 328 --- .../custom-values.yaml | 277 -- .../custom-values.yaml | 119 - .../custom-values.yaml | 497 ---- .../custom-values.yaml | 170 -- .../telegraf-operator/custom-values.yaml | 150 -- .../temporal/custom-values.yaml | 157 -- .../custom-values.yaml | 296 --- .../custom-values.yaml | 272 -- .../victoria-metrics-agent/custom-values.yaml | 296 --- .../custom-values.yaml | 304 --- .../custom-values.yaml | 340 --- .../custom-values.yaml | 305 --- .../custom-values.yaml | 340 --- .../victoria-metrics-alert/custom-values.yaml | 305 --- .../custom-values.yaml | 234 -- .../custom-values.yaml | 297 --- .../custom-values.yaml | 316 --- .../victoriametrics-agent/custom-values.yaml | 481 ---- .../victoriametrics-insert/custom-values.yaml | 411 --- .../victoriametrics-select/custom-values.yaml | 449 ---- .../custom-values.yaml | 363 --- .../conntrack-adjuster/custom-values.yaml | 26 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-external-1/custom-values.yaml | 86 - .../contour-external/custom-values.yaml | 86 - .../contour-internal-0/custom-values.yaml | 90 - .../contour-internal-1/custom-values.yaml | 91 - .../custom-values.yaml | 88 - .../custom-values.yaml | 89 - .../coredns/custom-values.yaml | 31 - .../external-secrets/custom-values.yaml | 50 - .../flagger/custom-values.yaml | 68 - .../fluentd/custom-values.yaml | 720 ------ .../ingress-nginx/custom-values.yaml | 32 - .../keda/custom-values.yaml | 35 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../custom-values.yaml | 119 - .../custom-values.yaml | 111 - .../custom-values.yaml | 498 ---- .../custom-values.yaml | 170 -- .../telegraf-operator/custom-values.yaml | 151 -- .../victoria-metrics-agent/custom-values.yaml | 298 --- .../custom-values.yaml | 235 -- .../custom-values.yaml | 301 --- .../custom-values.yaml | 317 --- .../k8s-dataengg-prd-ase1/README.md | 3 - .../alloy/custom-values.yaml | 50 - .../aurva-dataplane/custom-values.yaml | 675 ----- .../cert-manager/custom-values.yaml | 129 - .../conntrack-adjuster/custom-values.yaml | 19 - .../contour-ca-issuer/custom-values.yaml | 17 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-external/custom-values.yaml | 139 - .../contour-internal-0/custom-values.yaml | 89 - .../contour-internal-1/custom-values.yaml | 94 - .../custom-values.yaml | 87 - .../custom-values.yaml | 92 - .../coredns/custom-values.yaml | 28 - .../coroot-node-agent/custom-values.yaml | 53 - .../external-secrets/custom-values.yaml | 61 - .../flagger/custom-values.yaml | 63 - .../fluentd/custom-values.yaml | 730 ------ .../grafana/custom-values.yaml | 102 - .../ingress-nginx-external/custom-values.yaml | 34 - .../ingress-nginx-internal/custom-values.yaml | 33 - .../ingress-nginx-secured/custom-values.yaml | 33 - .../keda/custom-values.yaml | 61 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../kubectl-mcp-server/custom-values.yaml | 117 - .../kubernetes-dashboard/custom-values.yaml | 442 ---- .../kyverno/custom-values.yaml | 2240 ---------------- .../kyverno/policies/protect-namespace.yaml | 40 - .../kyverno/policies/restrict-replicas.yaml | 29 - .../loadtester/custom-values.yaml | 116 - .../custom-values.yaml | 284 --- .../custom-values.yaml | 134 - .../custom-values.yaml | 496 ---- .../custom-values.yaml | 170 -- .../telegraf-operator/custom-values.yaml | 179 -- .../custom-values.yaml | 296 --- .../custom-values.yaml | 272 -- .../custom-values.yaml | 296 --- .../victoria-metrics-agent/custom-values.yaml | 296 --- .../custom-values.yaml | 304 --- .../custom-values.yaml | 340 --- .../custom-values.yaml | 304 --- .../custom-values.yaml | 340 --- .../victoria-metrics-alert/custom-values.yaml | 305 --- .../custom-values.yaml | 234 -- .../custom-values.yaml | 234 -- .../custom-values.yaml | 293 --- .../custom-values.yaml | 293 --- .../custom-values.yaml | 316 --- .../custom-values.yaml | 316 --- .../victoriametrics-agent/custom-values.yaml | 485 ---- .../victoriametrics-insert/custom-values.yaml | 411 --- .../victoriametrics-select/custom-values.yaml | 447 ---- .../custom-values.yaml | 363 --- .../conntrack-adjuster/custom-values.yaml | 26 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-external/custom-values.yaml | 139 - .../contour-internal-0/custom-values.yaml | 91 - .../contour-internal-1/custom-values.yaml | 94 - .../custom-values.yaml | 89 - .../custom-values.yaml | 92 - .../coredns/custom-values.yaml | 30 - .../external-secrets/custom-values.yaml | 50 - .../flagger/custom-values.yaml | 69 - .../fluentd/custom-values.yaml | 720 ------ .../ingress-nginx-external/custom-values.yaml | 34 - .../ingress-nginx-internal/custom-values.yaml | 33 - .../ingress-nginx/custom-values.yaml | 32 - .../keda/custom-values.yaml | 61 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../custom-values.yaml | 119 - .../custom-values.yaml | 498 ---- .../custom-values.yaml | 170 -- .../telegraf-operator/custom-values.yaml | 151 -- .../victoria-metrics-agent/custom-values.yaml | 298 --- .../custom-values.yaml | 235 -- .../custom-values.yaml | 301 --- .../custom-values.yaml | 317 --- .../k8s-datascience-prd-ase1/README.md | 3 - .../alloy/custom-values.yaml | 50 - .../aurva-dataplane/custom-values.yaml | 675 ----- .../cert-manager/custom-values.yaml | 129 - .../computeclass/contour-internal-0-arm.yaml | 28 - .../computeclass/contour-internal-1-arm.yaml | 28 - .../contour-internal-dataproc-arm.yaml | 22 - .../contour-internal-intra-0-arm.yaml | 28 - .../contour-internal-intra-1-arm.yaml | 28 - .../computeclass/contour-shared-arm.yaml | 28 - .../g2-standard-16-l4-compute-class.yaml | 49 - .../g2-standard-4-l4-compute-class.yaml | 49 - .../g2-standard-8-l4-300gb-compute-class.yaml | 52 - .../g2-standard-8-l4-compute-class.yaml | 49 - .../conntrack-adjuster/custom-values.yaml | 20 - .../contour-ca-issuer/custom-values.yaml | 17 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-internal-0/custom-values.yaml | 117 - .../contour-internal-1/custom-values.yaml | 118 - .../custom-values.yaml | 112 - .../custom-values.yaml | 114 - .../custom-values.yaml | 115 - .../coredns/custom-values.yaml | 28 - .../coroot-node-agent/custom-values.yaml | 52 - .../external-secrets/custom-values.yaml | 61 - .../flagger/custom-values.yaml | 62 - .../fluentd/custom-values.yaml | 731 ------ .../ingress-nginx-internal/custom-values.yaml | 26 - .../keda/custom-values.yaml | 46 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../kubectl-mcp-server/custom-values.yaml | 117 - .../kubernetes-dashboard/custom-values.yaml | 442 ---- .../kyverno/custom-values.yaml | 2240 ---------------- .../kyverno/policies/protect-namespace.yaml | 40 - .../kyverno/policies/restrict-replicas.yaml | 29 - .../loadtester/custom-values.yaml | 116 - .../custom-values.yaml | 281 --- .../custom-values.yaml | 134 - .../paused-container/custom-values.yaml | 82 - .../custom-values.yaml | 496 ---- .../custom-values.yaml | 170 -- .../telegraf-operator/custom-values.yaml | 150 -- .../custom-values.yaml | 295 --- .../custom-values.yaml | 272 -- .../victoria-metrics-agent/custom-values.yaml | 295 --- .../custom-values.yaml | 304 --- .../custom-values.yaml | 340 --- .../custom-values.yaml | 305 --- .../custom-values.yaml | 340 --- .../victoria-metrics-alert/custom-values.yaml | 305 --- .../custom-values.yaml | 234 -- .../custom-values.yaml | 293 --- .../custom-values.yaml | 316 --- .../victoriametrics-agent/custom-values.yaml | 485 ---- .../victoriametrics-insert/custom-values.yaml | 411 --- .../victoriametrics-select/custom-values.yaml | 445 ---- .../custom-values.yaml | 363 --- .../alloy/custom-values.yaml | 50 - .../aurva-dataplane/custom-values.yaml | 652 ----- .../conntrack-adjuster/custom-values.yaml | 26 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-internal-0/custom-values.yaml | 94 - .../contour-internal-1/custom-values.yaml | 110 - .../custom-values.yaml | 92 - .../custom-values.yaml | 108 - .../coredns/custom-values.yaml | 30 - .../external-secrets/custom-values.yaml | 50 - .../flagger/custom-values.yaml | 63 - .../fluentd/custom-values.yaml | 718 ------ .../ingress-nginx-internal/custom-values.yaml | 26 - .../keda/custom-values.yaml | 46 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../kubernetes-dashboard/custom-values.yaml | 442 ---- .../loadtester/custom-values.yaml | 116 - .../custom-values.yaml | 281 --- .../custom-values.yaml | 135 - .../paused-container/custom-values.yaml | 82 - .../custom-values.yaml | 497 ---- .../custom-values.yaml | 171 -- .../telegraf-operator/custom-values.yaml | 151 -- .../victoria-metrics-agent/custom-values.yaml | 296 --- .../victoria-metrics-alert/custom-values.yaml | 305 --- .../custom-values.yaml | 235 -- .../custom-values.yaml | 299 --- .../custom-values.yaml | 317 --- helm-overrides/k8s-demand-prd-ase1/README.md | 3 - .../alloy/custom-values.yaml | 50 - .../aurva-dataplane/custom-values.yaml | 675 ----- .../cert-manager/custom-values.yaml | 129 - .../conntrack-adjuster/custom-values.yaml | 20 - .../contour-ca-issuer/custom-values.yaml | 17 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-external/custom-values.yaml | 85 - .../contour-internal-0/custom-values.yaml | 92 - .../contour-internal-1/custom-values.yaml | 89 - .../custom-values.yaml | 90 - .../custom-values.yaml | 87 - .../coredns/custom-values.yaml | 28 - .../coroot-node-agent/custom-values.yaml | 52 - .../external-secrets/custom-values.yaml | 61 - .../flagger/custom-values.yaml | 62 - .../fluentd/custom-values.yaml | 734 ------ .../keda/custom-values.yaml | 61 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../kubectl-mcp-server/custom-values.yaml | 117 - .../kyverno/custom-values.yaml | 2240 ---------------- .../kyverno/policies/protect-namespace.yaml | 40 - .../kyverno/policies/restrict-replicas.yaml | 29 - .../loadtester/custom-values.yaml | 116 - .../node-thp-config/custom-values.yaml | 9 - .../custom-values.yaml | 281 --- .../custom-values.yaml | 133 - .../custom-values.yaml | 496 ---- .../custom-values.yaml | 170 -- .../telegraf-operator/custom-values.yaml | 292 --- .../custom-values.yaml | 296 --- .../custom-values.yaml | 272 -- .../custom-values.yaml | 304 --- .../custom-values.yaml | 340 --- .../custom-values.yaml | 305 --- .../custom-values.yaml | 340 --- .../victoria-metrics-alert/custom-values.yaml | 305 --- .../custom-values.yaml | 234 -- .../custom-values.yaml | 293 --- .../custom-values.yaml | 316 --- .../custom-values.yaml | 479 ---- .../custom-values.yaml | 481 ---- .../victoriametrics-agent/custom-values.yaml | 485 ---- .../victoriametrics-insert/custom-values.yaml | 411 --- .../victoriametrics-select/custom-values.yaml | 449 ---- .../custom-values.yaml | 363 --- .../conntrack-adjuster/custom-values.yaml | 26 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-external/custom-values.yaml | 85 - .../contour-internal-0/custom-values.yaml | 97 - .../contour-internal-1/custom-values.yaml | 89 - .../custom-values.yaml | 95 - .../custom-values.yaml | 87 - .../coredns/custom-values.yaml | 30 - .../external-secrets/custom-values.yaml | 50 - .../flagger/custom-values.yaml | 69 - .../fluentd/custom-values.yaml | 720 ------ .../ingress-nginx/custom-values.yaml | 32 - .../keda/custom-values.yaml | 35 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../custom-values.yaml | 119 - .../custom-values.yaml | 498 ---- .../custom-values.yaml | 170 -- .../telegraf-operator/custom-values.yaml | 151 -- .../victoria-metrics-agent/custom-values.yaml | 298 --- .../custom-values.yaml | 235 -- .../custom-values.yaml | 301 --- .../custom-values.yaml | 317 --- .../k8s-dengspark-di-prd-ase1/README.md | 3 - .../conntrack-adjuster/custom-values.yaml | 26 - .../external-secrets/custom-values.yaml | 50 - .../flagger/custom-values.yaml | 55 - .../fluentd/custom-values.yaml | 764 ------ .../ingress-nginx-external/custom-values.yaml | 30 - .../ingress-nginx-internal/custom-values.yaml | 26 - .../keda/custom-values.yaml | 45 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../custom-values.yaml | 496 ---- .../victoria-metrics-agent/custom-values.yaml | 295 --- .../victoria-metrics-alert/custom-values.yaml | 304 --- .../custom-values.yaml | 234 -- .../custom-values.yaml | 293 --- .../custom-values.yaml | 316 --- .../victoriametrics-agent/custom-values.yaml | 319 --- .../victoriametrics-insert/custom-values.yaml | 313 --- .../victoriametrics-select/custom-values.yaml | 331 --- .../custom-values.yaml | 346 --- .../k8s-dengspark-notebook-prd-ase1/README.md | 3 - .../external-secrets/custom-values.yaml | 50 - .../flagger/custom-values.yaml | 55 - .../fluentd/custom-values.yaml | 733 ------ .../ingress-nginx-external/custom-values.yaml | 30 - .../ingress-nginx-internal/custom-values.yaml | 31 - .../keda/custom-values.yaml | 38 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../custom-values.yaml | 496 ---- .../rancher/custom-values.yaml | 33 - .../victoria-metrics-agent/custom-values.yaml | 295 --- .../victoria-metrics-alert/custom-values.yaml | 304 --- .../custom-values.yaml | 234 -- .../custom-values.yaml | 293 --- .../custom-values.yaml | 316 --- .../k8s-dengspark-prd-ase1/README.md | 3 - .../conntrack-adjuster/custom-values.yaml | 26 - .../external-secrets/custom-values.yaml | 50 - .../flagger/custom-values.yaml | 55 - .../fluentd/custom-values.yaml | 764 ------ .../ingress-nginx-external/custom-values.yaml | 30 - .../ingress-nginx-internal/custom-values.yaml | 26 - .../keda/custom-values.yaml | 45 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../custom-values.yaml | 496 ---- .../victoria-metrics-agent/custom-values.yaml | 295 --- .../custom-values.yaml | 340 --- .../victoria-metrics-alert/custom-values.yaml | 304 --- .../custom-values.yaml | 234 -- .../custom-values.yaml | 295 --- .../custom-values.yaml | 316 --- .../victoriametrics-agent/custom-values.yaml | 314 --- .../victoriametrics-insert/custom-values.yaml | 311 --- .../victoriametrics-select/custom-values.yaml | 332 --- .../custom-values.yaml | 345 --- .../k8s-dscispark-prd-ase1/README.md | 3 - .../conntrack-adjuster/custom-values.yaml | 26 - .../external-secrets/custom-values.yaml | 50 - .../flagger/custom-values.yaml | 55 - .../fluentd/custom-values.yaml | 764 ------ .../ingress-nginx-internal/custom-values.yaml | 26 - .../keda/custom-values.yaml | 38 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../custom-values.yaml | 496 ---- .../victoria-metrics-agent/custom-values.yaml | 295 --- .../custom-values.yaml | 340 --- .../victoria-metrics-alert/custom-values.yaml | 304 --- .../custom-values.yaml | 234 -- .../custom-values.yaml | 295 --- .../custom-values.yaml | 316 --- .../victoriametrics-agent/custom-values.yaml | 317 --- .../victoriametrics-insert/custom-values.yaml | 267 -- .../victoriametrics-select/custom-values.yaml | 331 --- .../custom-values.yaml | 345 --- helm-overrides/k8s-dsgpu-prd-ase1/README.md | 3 - .../aurva-dataplane/custom-values.yaml | 660 ----- .../cert-manager/custom-values.yaml | 129 - .../g2-standard-16-l4-compute-class.yaml | 37 - .../g2-standard-4-l4-compute-class.yaml | 37 - .../g2-standard-8-l4-300gb-compute-class.yaml | 39 - .../g2-standard-8-l4-compute-class.yaml | 37 - .../conntrack-adjuster/custom-values.yaml | 26 - .../contour-ca-issuer/custom-values.yaml | 17 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-internal-0/custom-values.yaml | 102 - .../contour-internal-1/custom-values.yaml | 112 - .../custom-values.yaml | 95 - .../custom-values.yaml | 110 - .../coredns/custom-values.yaml | 28 - ...u-prd-envoy-headless-external-dns-svc.yaml | 30 - .../external-dns/custom-values.yaml | 60 - .../external-secrets/custom-values.yaml | 61 - .../flagger/custom-values.yaml | 62 - .../fluentd/custom-values.yaml | 731 ------ .../ingress-nginx-internal/custom-values.yaml | 26 - .../keda/custom-values.yaml | 46 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 148 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../kubectl-mcp-server/custom-values.yaml | 121 - .../loadtester/custom-values.yaml | 116 - .../custom-values.yaml | 281 --- .../custom-values.yaml | 134 - .../paused-container/custom-values.yaml | 82 - .../custom-values.yaml | 496 ---- .../custom-values.yaml | 170 -- .../telegraf-operator/custom-values.yaml | 150 -- .../custom-values.yaml | 340 --- .../victoriametrics-agent/custom-values.yaml | 479 ---- .../victoriametrics-insert/custom-values.yaml | 411 --- .../victoriametrics-select/custom-values.yaml | 444 ---- .../custom-values.yaml | 363 --- helm-overrides/k8s-farmiso-prd-ase1/README.md | 3 - .../alloy/custom-values.yaml | 57 - .../aurva-dataplane/custom-values.yaml | 673 ----- .../cert-manager/custom-values.yaml | 129 - .../conntrack-adjuster/custom-values.yaml | 17 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-external/custom-values.yaml | 101 - .../contour-internal-0/custom-values.yaml | 104 - .../custom-values.yaml | 101 - .../coredns/custom-values.yaml | 28 - .../external-secrets/custom-values.yaml | 61 - .../flagger/custom-values.yaml | 62 - .../fluentd/custom-values.yaml | 812 ------ .../keda/custom-values.yaml | 26 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../kubectl-mcp-server/custom-values.yaml | 119 - .../kyverno/custom-values.yaml | 2240 ---------------- .../kyverno/policies/protect-namespace.yaml | 40 - .../kyverno/policies/restrict-replicas.yaml | 29 - .../loadtester/custom-values.yaml | 116 - .../custom-values.yaml | 281 --- .../custom-values.yaml | 134 - .../custom-values.yaml | 491 ---- .../custom-values.yaml | 170 -- .../telegraf-operator/custom-values.yaml | 151 -- .../custom-values.yaml | 296 --- .../custom-values.yaml | 272 -- .../custom-values.yaml | 304 --- .../custom-values.yaml | 340 --- .../victoria-metrics-alert/custom-values.yaml | 305 --- .../victoriametrics-agent/custom-values.yaml | 479 ---- .../victoriametrics-insert/custom-values.yaml | 411 --- .../victoriametrics-select/custom-values.yaml | 448 ---- .../custom-values.yaml | 363 --- .../k8s-ml-platform-prd-ase1/README.md | 3 - .../cert-manager/custom-values.yaml | 129 - .../conntrack-adjuster/custom-values.yaml | 26 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-internal-0/custom-values.yaml | 94 - .../contour-internal-1/custom-values.yaml | 110 - .../coredns/custom-values.yaml | 27 - .../external-secrets/custom-values.yaml | 50 - .../flagger/custom-values.yaml | 62 - .../fluentd/custom-values.yaml | 717 ------ .../keda/custom-values.yaml | 46 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../custom-values.yaml | 133 - .../custom-values.yaml | 496 ---- .../custom-values.yaml | 170 -- .../telegraf-operator/custom-values.yaml | 150 -- .../victoria-metrics-agent/custom-values.yaml | 295 --- .../custom-values.yaml | 340 --- .../victoria-metrics-alert/custom-values.yaml | 304 --- .../custom-values.yaml | 234 -- .../custom-values.yaml | 293 --- .../custom-values.yaml | 316 --- .../coredns/custom-values.yaml | 26 - .../deepfence-console/custom-values.yaml | 534 ---- .../deepfence-router/custom-values.yaml | 175 -- .../external-secrets/custom-values.yaml | 45 - .../flagger/custom-values.yaml | 54 - .../ingress-nginx/custom-values.yaml | 15 - .../keda/custom-values.yaml | 18 - .../kube-dns/custom-values.yaml | 2 - .../kube-state-metrics/custom-values.yaml | 478 ---- .../custom-values.yaml | 491 ---- .../custom-values.yaml | 170 -- .../telegraf-operator/custom-values.yaml | 152 -- .../victoria-metrics-agent/custom-values.yaml | 295 --- .../custom-values.yaml | 228 -- .../custom-values.yaml | 290 --- .../custom-values.yaml | 317 --- .../bifrost/custom-values.yaml | 281 --- .../cert-manager/custom-values.yaml | 129 - .../cc-contour-vmagent-od-compute-class.yaml | 21 - ...c-shared-cost-optimised-compute-class.yaml | 55 - .../cc-shared-devops-od-compute-class.yaml | 21 - .../cc-shared-gpu-a2-compute-class.yaml | 25 - .../cc-shared-gpu-a3-compute-class.yaml | 25 - .../cc-starrocks-n2-highmem-16.yaml | 16 - .../cc-starrocks-n2-highmem-32.yaml | 16 - .../cc-starrocks-n2-highmem-8.yaml | 16 - .../conntrack-adjuster/custom-values.yaml | 25 - .../contour-ca-issuer/custom-values.yaml | 17 - .../contour-cert-checker/custom-values.yaml | 37 - .../contour-external/custom-values.yaml | 115 - .../contour-internal-0/custom-values.yaml | 125 - .../contour-internal-1/custom-values.yaml | 116 - .../custom-values.yaml | 109 - .../coredns/custom-values.yaml | 38 - .../deepgram-onprem/custom-values.yaml | 897 ------- .../external-secrets/custom-values.yaml | 50 - .../fluentd/custom-values.yaml | 835 ------ .../ingress-nginx-internal/custom-values.yaml | 42 - .../keda/custom-values.yaml | 98 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 486 ---- .../kubectl-mcp-server/custom-values.yaml | 120 - .../kyverno/custom-values.yaml | 2240 ---------------- .../kyverno/policies/protect-namespace.yaml | 39 - .../policies/request-limit-policy.yaml | 31 - .../kyverno/policies/restrict-replicas.yaml | 27 - .../custom-values.yaml | 112 - .../custom-values.yaml | 504 ---- .../custom-values.yaml | 177 -- .../telegraf-operator/custom-values.yaml | 240 -- .../victoria-metrics-agent/custom-values.yaml | 304 --- .../victoriametrics-agent/custom-values.yaml | 474 ---- .../deepgram-onprem/custom-values.yaml | 795 ------ helm-overrides/k8s-supply-prd-ase1/README.md | 3 - .../alloy/custom-values.yaml | 53 - .../aurva-dataplane/custom-values.yaml | 676 ----- .../cert-manager/custom-values.yaml | 129 - .../conntrack-adjuster/custom-values.yaml | 20 - .../contour-ca-issuer/custom-values.yaml | 17 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-external/custom-values.yaml | 86 - .../contour-internal-0/custom-values.yaml | 99 - .../contour-internal-1/custom-values.yaml | 92 - .../custom-values.yaml | 97 - .../custom-values.yaml | 90 - .../coredns/custom-values.yaml | 28 - .../coroot-node-agent/custom-values.yaml | 54 - .../deepgram-onprem-v2/custom-values.yaml | 410 --- .../deepgram-onprem/custom-values.yaml | 918 ------- .../external-secrets/custom-values.yaml | 61 - .../flagger/custom-values.yaml | 63 - .../fluentd-sumoduolite-np/custom-values.yaml | 724 ------ .../fluentd/custom-values.yaml | 855 ------- .../keda/custom-values.yaml | 45 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../kubectl-mcp-server/custom-values.yaml | 117 - .../kubernetes-dashboard/custom-values.yaml | 442 ---- .../kyverno/custom-values.yaml | 2240 ---------------- .../kyverno/policies/protect-namespace.yaml | 40 - .../kyverno/policies/restrict-replicas.yaml | 29 - .../loadtester/custom-values.yaml | 116 - .../node-thp-config/custom-values.yaml | 9 - .../custom-values.yaml | 288 --- .../custom-values.yaml | 110 - .../custom-values.yaml | 135 - .../custom-values.yaml | 496 ---- .../custom-values.yaml | 170 -- .../custom-values.yaml | 150 -- .../telegraf-operator/custom-values.yaml | 224 -- .../custom-values.yaml | 295 --- .../custom-values.yaml | 272 -- .../victoria-metrics-agent/custom-values.yaml | 298 --- .../custom-values.yaml | 304 --- .../custom-values.yaml | 332 --- .../custom-values.yaml | 305 --- .../custom-values.yaml | 340 --- .../victoria-metrics-alert/custom-values.yaml | 305 --- .../custom-values.yaml | 305 --- .../custom-values.yaml | 1377 ---------- .../custom-values.yaml | 234 -- .../custom-values.yaml | 293 --- .../custom-values.yaml | 293 --- .../custom-values.yaml | 316 --- .../victoriametrics-agent/custom-values.yaml | 485 ---- .../custom-values.yaml | 409 --- .../victoriametrics-insert/custom-values.yaml | 411 --- .../custom-values.yaml | 451 ---- .../victoriametrics-select/custom-values.yaml | 449 ---- .../custom-values.yaml | 363 --- .../custom-values.yaml | 363 --- .../conntrack-adjuster/custom-values.yaml | 26 - .../contour-cert-checker/custom-values.yaml | 28 - .../contour-external/custom-values.yaml | 86 - .../contour-internal-0/custom-values.yaml | 97 - .../contour-internal-1/custom-values.yaml | 89 - .../custom-values.yaml | 97 - .../custom-values.yaml | 87 - .../coredns/custom-values.yaml | 30 - .../external-secrets/custom-values.yaml | 50 - .../flagger/custom-values.yaml | 69 - .../fluentd/custom-values.yaml | 720 ------ .../ingress-nginx/custom-values.yaml | 32 - .../keda/custom-values.yaml | 35 - .../kube-dns/custom-values.yaml | 2 - .../kube-events/custom-values.yaml | 149 -- .../kube-state-metrics/custom-values.yaml | 478 ---- .../custom-values.yaml | 119 - .../custom-values.yaml | 498 ---- .../custom-values.yaml | 170 -- .../telegraf-operator/custom-values.yaml | 151 -- .../victoria-metrics-agent/custom-values.yaml | 298 --- .../custom-values.yaml | 235 -- .../custom-values.yaml | 301 --- .../custom-values.yaml | 317 --- index.md | 123 - .../jenkins-filestore-caching/dev/pv.yaml | 20 - .../jenkins-filestore-caching/dev/pvc.yaml | 13 - manifests/jfrog-filestore-data/dev/pv.yaml | 21 - manifests/jfrog-filestore-data/dev/pvc.yaml | 13 - .../nginx-static-content-manifest.yaml | 213 -- .../demand/nginx-static-content-manifest.yaml | 213 -- .../nginx-static-content-manifest.yaml | 213 -- .../supply/nginx-static-content-manifest.yaml | 213 -- .../priorityclass-high.yaml | 8 - .../priorityclass-low.yaml | 8 - .../priorityclass-high.yaml | 8 - .../priorityclass-low.yaml | 8 - .../priorityclass-high.yaml | 8 - .../priorityclass-low.yaml | 8 - .../spot-termination-handler.yaml | 8 - .../priorityclass-high.yaml | 8 - .../priorityclass-low.yaml | 8 - .../priorityclass-high.yaml | 8 - .../priorityclass-high.yaml | 8 - .../priorityclass-low.yaml | 8 - .../priorityclass-high.yaml | 8 - .../priorityclass-low.yaml | 8 - .../priorityclass-high.yaml | 8 - .../priorityclass-low.yaml | 8 - .../priorityclass-high.yaml | 8 - .../priorityclass-low.yaml | 8 - .../priorityclass-high.yaml | 8 - .../priorityclass-low.yaml | 8 - .../priorityclass-high.yaml | 8 - .../k8s-sec-admin-ase1/priorityclass-low.yaml | 8 - .../priorityclass-high.yaml | 8 - .../priorityclass-low.yaml | 8 - .../hyperdisk-balanced.yaml | 10 - .../pd-standard-retain.yaml | 10 - .../pulse-nfs-sc-prd.yaml | 10 - .../gke-central-prd-ase1a/sc-pd-ssd.yaml | 12 - .../gke-central-prd-ase1a/sc-pd-standard.yaml | 12 - .../enterprise-multishare-rwx.yaml | 20 - .../enterprise-rwx.yaml | 18 - .../hyperdisk-balanced.yaml | 18 - .../pd-standard-retain.yaml | 11 - .../premium-rwx.yaml | 18 - .../pulse-nfs-sc-prd.yaml | 9 - .../pulse-nfs-sc-secured-prd.yaml | 9 - .../regional-rwx.yaml | 18 - .../sc-filestore-standard.yaml | 12 - .../standard-rwx-retain.yaml | 18 - .../standard-rwx.yaml | 18 - .../gke-datascience-prd-ase1a/zonal-rwx.yaml | 18 - post-commit-scripts/._commit-metric.sh | Bin 311 -> 0 bytes post-commit-scripts/._runner.sh | Bin 212 -> 0 bytes post-commit-scripts/commit-metric.sh | 1551 ------------ post-commit-scripts/runner.sh | 47 - pre-commit-scripts/._cac-validate.sh | Bin 212 -> 0 bytes pre-commit-scripts/._runner.sh | Bin 212 -> 0 bytes pre-commit-scripts/._trufflehog-hook.sh | Bin 212 -> 0 bytes pre-commit-scripts/._yaakhook.sh | Bin 212 -> 0 bytes pre-commit-scripts/cac-validate.sh | 60 - pre-commit-scripts/runner.sh | 47 - pre-commit-scripts/trufflehog-hook.sh | 56 - pre-commit-scripts/yaakhook.sh | 30 - repository.yaml | 5 - skills/infra/add-infra-tool.md | 252 -- skills/infra/bump-chart-version.md | 157 -- skills/infra/check-cluster-health.md | 249 -- skills/infra/diagnose-deployment.md | 224 -- skills/infra/diagnose-scheduling.md | 174 -- skills/infra/onboard-app.md | 203 -- .../ADR-A1-cache-vs-upstream-charts.md | 71 - .../ADR-A2-blue-green-sibling-pattern.md | 83 - .../analyses/ADR-A3-per-cluster-scheduling.md | 78 - ...raw-manifest-sidecars-in-helm-overrides.md | 75 - .../ADR-A5-manual-sync-default-for-infra.md | 66 - wiki/entities/DevOps Infra Helm Charts.md | 214 -- 1480 files changed, 61 insertions(+), 273821 deletions(-) delete mode 100644 CLAUDE.md delete mode 100644 claude/00-overview.md delete mode 100644 claude/01-repo-structure.md delete mode 100644 claude/02-cluster-fleet.md delete mode 100644 claude/03-chart-inventory.md delete mode 100644 claude/04-override-hierarchy.md delete mode 100644 claude/05-deploy-lifecycle.md delete mode 100644 claude/06-secrets-and-identity.md delete mode 100644 claude/07-singletons-and-blast-radius.md delete mode 100644 claude/08-pre-commit-and-hooks.md delete mode 100644 claude/09-common-tasks.md delete mode 100644 claude/10-glossary-and-references.md delete mode 100644 contour-nodeselector-tolerations-summary.md delete mode 100644 docs/architecture.md delete mode 100644 docs/global/AGENT_BOUNDARIES.md delete mode 100644 docs/global/SANCTITY_RULES.md delete mode 100644 docs/global/agent-operations-guide.md delete mode 100644 docs/global/coding-guidelines/argocd.md delete mode 100644 docs/global/coding-guidelines/helm-values.md delete mode 100644 docs/global/coding-guidelines/observability.md delete mode 100644 docs/global/escalation-matrix.md delete mode 100644 docs/platform/procedures/add-contour-route.md delete mode 100644 docs/platform/procedures/blue-green-chart-migration.md delete mode 100644 docs/platform/procedures/deboard-app.md delete mode 100644 docs/platform/procedures/fork-upstream-chart.md delete mode 100644 docs/platform/procedures/modify-alert-rules.md delete mode 100644 docs/platform/procedures/modify-observability-config.md delete mode 100644 docs/platform/procedures/onboard-app-to-cluster.md delete mode 100644 docs/platform/procedures/onboard-new-cluster.md delete mode 100644 docs/platform/procedures/update-chart-version.md delete mode 100644 docs/platform/runbooks/argocd-sync-failure.md delete mode 100644 docs/platform/runbooks/ingress-down.md delete mode 100644 docs/platform/runbooks/metrics-gap.md delete mode 100644 docs/platform/runbooks/pod-pending-scheduling.md delete mode 100644 docs/platform/runbooks/vault-unavailable.md delete mode 100644 docs/platform/schemas/custom-values-schema.md delete mode 100644 docs/platform/schemas/incubator-values-schema.md delete mode 100644 docs/platform/schemas/raw-manifest-sidecar-schema.md delete mode 100644 docs/platform/schemas/storageclass-priorityclass-schema.md delete mode 100644 helm-overrides/db-2516183257845181-c-1204-195038-428/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/db-2516183257845181-c-1204-195038-428/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/db-3474993620361343-3-1215-134240-999/ingress-nginx-internal/custom-values.yaml delete mode 100644 helm-overrides/db-3474993620361343-3-1215-134240-999/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/db-3474993620361343-3-1215-134240-999/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/db-3474993620361343-3-1215-134240-999/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/db-3474993620361343-3-1215-134240-999/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/db-3474993620361343-3-1215-134240-999/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/db-5070361081051941-0-0108-201559-719/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/db-5070361081051941-0-0108-201559-719/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/db-8070720513218850-d-1128-192026-437/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/db-8070720513218850-d-1128-192026-437/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/db-8405596239050069-0-0222-204522-227/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/db-8405596239050069-0-0222-204522-227/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/README.md delete mode 100644 helm-overrides/gke-central-prd-ase1a/ai-gateway-ext/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/ai-gateway/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/akamai-observability-mcp/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/aurva-dataplane/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/cert-manager/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/clickhouse/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/azul-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/central-devops-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/central-kyverno-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/compactduo-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/compacttetra-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/contour-external-arm.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/contour-external-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-0-arm.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-0-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-1-arm.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-1-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-intra-0-arm.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-intra-0-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-intra-1-arm.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-intra-1-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/contour-shared-arm.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/contour-shared-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/devops-mcp-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/gatekeeper-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/loghouse-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/megaduo-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/megaduolite-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/megaoctalite-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/megatetra-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/megatetralite-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/megauno-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/megaunolite-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/mlp-g2-standard-8-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/sale-rescue-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/session-mgr-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/sumoduo-c4d-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/sumoduo-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/sumoduolite-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/sumotetra-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/sumouno-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/sumounolite-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/vmagent-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/vmagent-dr-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/vmagent-mds-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/vminsert-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/vminsert-mds-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/vmselect-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/vmselect-mds-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-mds-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-sale-24aug-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-sale-cc.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/contour-ca-issuer/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/contour-external-1/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/contour-external/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/coredns/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/elasticsearch-mcp/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/etcd/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/fireworks-ai/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/flagger/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/fluentd-copy/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/fluentd/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/keda/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/kube-events/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/kubectl-mcp-server/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/kyverno/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/kyverno/policies/protect-namespace.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/kyverno/policies/restrict-replicas.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/loadtester/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/opentelemetry-claude-metrics/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/opentelemetry-codex-metrics/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/opentelemetry-coralogix/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/temporal/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/victoria-metrics-alert-secured-stateful/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/victoriametrics-insert/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/victoriametrics-select/custom-values.yaml delete mode 100644 helm-overrides/gke-central-prd-ase1a/victoriametrics-storage/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/README.md delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/aurva-dataplane/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/cert-manager/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/16c-104g-dp-dpcon-tco-farm-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/16c-104g-farm-dp-dpcon-tworker-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/52c-150g-deng-dpexp-ab-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/8c-64g-dp-dpnrt-druid-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/8c-64g-dp-dpnrt-druid-od-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/alluxio-32c-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/alluxio-hmem32-8-lssd-prd-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/c3d-hmem-8-sp-dpnrt-druid-int-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/compactocta-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/compacttetra-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-external-cc-v1.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-external-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-internal-0-cc-v1.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-internal-0-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-internal-1-cc-v1.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-internal-1-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-intra-0-cc-v1.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-intra-1-cc-v1.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-intra-1-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-shared-cc-v1.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-shared-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dataengg-devops.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dp-airflow-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dp-dpnrt-zookeeper-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-alluxio-n2-hmem-8-c-od.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-32-a-co-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-32-b-co-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-32-c-co-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-a-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-a-co-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-b-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-b-co-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-c-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-c-co-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-c-w-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-16-b-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-32-b-od-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-48-a-sensitive-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-48-b-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-48-c-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/kuberay-operator-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/megaduo-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/megaduolite-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/megaquad-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/megatetra-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/n2-hmem-64-ondemand-a-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/n2d-highmem-8-dp-dpcon-zep-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/n2d-hmem-16-od-ls-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/n2d-hmem-8-sp-ls-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/n4-hmem-48-spot-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/nginx-shared-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/np-dp-kuberay-4c-16g-prd-ase1-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/os-trino-poc-n2d-dp-dpcon-wk-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/strimzi-kafka-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/sumoduo-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/sumoduolite-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/sumounolite-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/vmagent-mds-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/vminsert-mds-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/vmselect-mds-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/vmstack-startree-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/warpstream-dp-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/computeclass/wk-trino-spot-48c-384g-prd-ase1-cc.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/contour-ca-issuer/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/contour-external/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/coredns/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/flagger/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/fluentd/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/ingress-nginx-external/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/ingress-nginx-internal/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/ingress-nginx-secured/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/keda/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/kube-events/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/kubectl-mcp-server/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/kyverno/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/kyverno/policies/protect-namespace.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/kyverno/policies/restrict-replicas.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/loadtester/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/opentelemetry-coralogix/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-alert-startree-stateful/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-insert-startree/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-select-startree/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-storage-startree/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/victoriametrics-insert/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/victoriametrics-select/custom-values.yaml delete mode 100644 helm-overrides/gke-dataengg-prd-ase1a/victoriametrics-storage/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/README.md delete mode 100644 helm-overrides/gke-datascience-prd-as1a/alloy/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/aurva-dataplane/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/cert-manager/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/computeclass/c3-highcpu-44-compute-class.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/computeclass/c3d-highcpu-30-compute-class.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/computeclass/c4a-highcpu-16-compute-class.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-0-arm.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-1-arm.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-dataproc-arm.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-intra-0-arm.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-intra-1-arm.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/computeclass/contour-shared-arm.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/computeclass/datascience-devops.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/computeclass/g2-standard-16-l4-compute-class.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/computeclass/g2-standard-4-l4-compute-class.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/computeclass/g2-standard-8-l4-300gb-compute-class.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/computeclass/g2-standard-8-l4-compute-class.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/computeclass/n2d-standard-48-4lssd-compute-class.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/contour-ca-issuer/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/contour-internal-dataproc/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/coredns/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/coroot-node-agent/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/flagger/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/fluentd/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/ingress-nginx-internal/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/keda/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/kube-events/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/kubectl-mcp-server/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/kubernetes-dashboard/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/kyverno/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/kyverno/policies/protect-namespace.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/kyverno/policies/restrict-replicas.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/loadtester/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/opentelemetry-coralogix/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/paused-container/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/victoria-metrics-agent-dr/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/victoria-metrics-agent-fb/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert-dr/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert-secured-stateful/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert-secured/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert-stateful/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/victoriametrics-insert/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/victoriametrics-select/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-as1a/victoriametrics-storage/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/README.md delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/alloy/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/aurva-dataplane/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/cert-manager/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/c3-standard-22-lssd-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-internal-0-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-internal-1-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-internal-dataproc-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-intra-0-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-intra-1-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-shared-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/datascience-devops-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/datascience-kyverno-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/ds-airflow-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/g2-l4-priority-class-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/g2-standard-4-l4-compute-class-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/g2-standard-8-l4-300gb-compute-class-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/g2-standard-8-l4-compute-class-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/load-testing-v2-mlp-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/megaduo-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/megaduo-ctz-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/megaduo-rto-consumer-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/megaduolite-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/megaquad-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/megatetra-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/megatetralite-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-a2-highgpu-1-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-c3d-hc-30-300-v1-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-c3d-highcpu-30-sz-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-c3d-highcpu-30-v1-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-c4d-localssd-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-g2-standard-16-v2-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-g2-standard-32-custom-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-g2-standard-8-zone-a-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/n2d-standard-48-4lssd-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/nginx-internal-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/np-dsci-ml-g2-standard-8-prd-ase1-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/rockdb-localssd-1-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/rust-onboard-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/ssd-cosmos-v2-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/sumoduo-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/sumotetra-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/vmagent-dr-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/vmagent-mds-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/vmagent-n4-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/vmagent-n4d-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/vminsert-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/vminsert-mds-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/vmselect-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/vmselect-mds-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/vmstorage-c4d-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/vmstorage-n4-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/computeclass/vmstorage-sale-24aug-cc.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/contour-ca-issuer/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/contour-internal-dataproc/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/coredns/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/coroot-node-agent/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/flagger/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/fluentd/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/ingress-nginx-internal/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/keda/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/kube-events/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/kubectl-mcp-server/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/kubernetes-dashboard/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/kyverno/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/kyverno/policies/protect-namespace.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/kyverno/policies/restrict-replicas.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/loadtester/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-alert-secured-stateful-v0/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-alert-stateful-v0/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/victoriametrics-agent-v0/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/victoriametrics-insert-v0/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/victoriametrics-select-v0/custom-values.yaml delete mode 100644 helm-overrides/gke-datascience-prd-ase1a/victoriametrics-storage-v0/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/README.md delete mode 100644 helm-overrides/gke-demand-prd-ase1a/alloy/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/aurva-dataplane/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/cert-manager/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/azul-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/compactduo-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/compacttetra-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/contour-external-cc-v1.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/contour-external-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/contour-internal-0-cc-v1.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/contour-internal-0-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/contour-internal-1-cc-v1.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/contour-internal-1-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/contour-intra-0-cc-v1.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/contour-intra-0-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/contour-intra-1-cc-v1.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/contour-intra-1-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/contour-shared-cc-v1.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/contour-shared-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/demand-devops-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/demand-kyverno-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/dmnd-c4a-32c64g-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/dns-coldstart-probe-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/megaduo-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/megaduo-spp-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/megaduolite-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/megatetra-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/megatetra-trnst-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/megatetralite-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/pdp-relay-v1-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/prod-comms-consumer-notification-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/prod-comms-consumer-v1-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/search-relay-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/sumoduo-azul-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/sumoduo-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/sumoduolite-azul-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/sumoduolite-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/sumoocta-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/sumotetra-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/sumotetralite-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/vm-agent-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/vmagent-c4d-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/vminsert-mds-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/vmselect-mds-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/contour-ca-issuer/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/contour-external/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/coredns/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/coroot-node-agent/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/flagger/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/fluentd/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/keda/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/kube-events/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/kubectl-mcp-server/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/kyverno/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/kyverno/policies/protect-namespace.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/kyverno/policies/restrict-replicas.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/loadtester/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/node-thp-config/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/opentelemetry-coralogix/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/victoria-metrics-alert-secured-stateful/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/victoriametrics-insert/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/victoriametrics-select/custom-values.yaml delete mode 100644 helm-overrides/gke-demand-prd-ase1a/victoriametrics-storage/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/README.md delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/aurva-dataplane/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/cert-manager/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/a2-highgpu-1-imgcache-compute-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3-highcpu-22-contour-internal-0-compute-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3-highcpu-22-contour-internal-1-compute-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3-highcpu-4-nginx-internal-compute-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3-highcpu-44-vmselect-compute-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3d-highcpu-8-contour-internal-0-compute-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3d-highcpu-8-contour-internal-1-compute-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c4d-highcpu-16-contour-internal-0-compute-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c4d-highcpu-16-contour-internal-1-compute-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/datascience-devops.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/dsgpu-kyverno-cc.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/e2-standard-4-datascience-devops-compute-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-l4-priority-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-16-l4-compute-class-ld.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-16-l4-compute-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-4-l4-compute-class-driver-latest.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-4-l4-compute-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-8-l4-300gb-compute-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-8-l4-compute-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-std-16-l4-priority-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/n4-highcpu-16-vminsert-compute-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/n4d-highcpu-64-vmagent-compute-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/computeclass/n4d-highmem-48-vmstorage-compute-class.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/contour-ca-issuer/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/coredns/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/external-dns-services/contour-internal-0-dsgpu-prd-envoy-headless-external-dns-svc.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/external-dns/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/flagger/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/fluentd/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/ingress-nginx-internal/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/keda/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/kube-events/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/kubectl-mcp-server/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/kyverno/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/kyverno/policies/protect-namespace.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/kyverno/policies/restrict-replicas.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/loadtester/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/opentelemetry-coralogix/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/paused-container/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/victoriametrics-insert/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/victoriametrics-select/custom-values.yaml delete mode 100644 helm-overrides/gke-dsgpu-prd-ase1a/victoriametrics-storage/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/README.md delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/aurva-dataplane/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/cert-manager/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/computeclass/compacttetra-cc.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-external-cc-v1.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-internal-0-cc-v1.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-intra-0-cc-v1.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-shared-cc-v1.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-shared-cc.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/computeclass/farmiso-devops.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/computeclass/megaquad-cc.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/computeclass/vmagent-mds-cc.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/computeclass/vminsert-mds-cc.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/computeclass/vmselect-mds-cc.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/contour-external/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/coredns/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/flagger/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/fluentd/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/keda/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/kube-events/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/kubectl-mcp-server/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/kyverno/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/kyverno/policies/protect-namespace.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/kyverno/policies/restrict-replicas.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/loadtester/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/opentelemetry-coralogix/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/victoriametrics-insert/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/victoriametrics-select/custom-values.yaml delete mode 100644 helm-overrides/gke-farmiso-prd-ase1a/victoriametrics-storage/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/README.md delete mode 100644 helm-overrides/gke-supply-prd-ase1a/alloy/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/aurva-dataplane/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/cert-manager/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/compactduo-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/compactocta-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/compacttetra-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/contour-external-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/contour-internal-0-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/contour-internal-1-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/contour-internal-1-new-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/contour-intra-0-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/contour-intra-1-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/contour-shared-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/deepgram-api-pool-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/deepgram-proxy-pool-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/dg-eg-pool-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/efficient-ai-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/exp-cx-gen-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/g2-standard-4-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/megaduo-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/megaduolite-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/megaoctalite-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/megatetra-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/megatetralite-azul-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/megatetralite-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/megauno-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/megaunolite-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/np-trino-n2-hmem-32-a-co-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/np-trino-n2-hmem-32-a-sp-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/sumoduo-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/sumoduo-cpu-mngr-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/sumoduolite-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/sumoduolite-op-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/sumoocta-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/sumotetra-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/sumotetra-taxonomy-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/sumotetra-trnst-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/sumotetralite-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/sumouno-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/sumounolite-azul-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/sumounolite-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/supply-devops-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/supply-kyverno-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/vm-stack-ht-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/vmagent-c4-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/vmagent-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/vmagent-mds-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/vmagent-n4-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/vminsert-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/vminsert-mds-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/vmselect-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/vmselect-mds-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-c4d-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-mds-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-n4-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-sale-24aug-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-tmp-cc.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/contour-ca-issuer/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/contour-external/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/coredns/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/coroot-node-agent/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/deepgram-onprem-v2/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/deepgram-onprem/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/flagger/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/fluentd/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/keda/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/kube-events/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/kubectl-mcp-server/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/kubernetes-dashboard/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/kyverno/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/kyverno/policies/protect-namespace.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/kyverno/policies/restrict-replicas.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/loadtester/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/node-thp-config/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/opentelemetry-coralogix/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/opentelemetry-daemonset-medium-np/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/telegraf-operator-custom/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/victoria-metrics-alert-secured-stateful/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/victoriametrics-insert/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/victoriametrics-select/custom-values.yaml delete mode 100644 helm-overrides/gke-supply-prd-ase1a/victoriametrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-aurva-prd-ase1/contour-internal/custom-values.yaml delete mode 100644 helm-overrides/k8s-aurva-prd-ase1/rancher/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/README.md delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/backend-nginx-cluster/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/cert-manager/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/ds-kafka-clusters-nginx/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/ingress-nginx-internal-v1/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/ingress-nginx-internal/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/nginx-demand/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/nginx-supply/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-alert/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/victoriametrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/victoriametrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-mqkafka-prd-ase1/victoriametrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/contour-external/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/coredns/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-alert/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/README.md delete mode 100644 helm-overrides/k8s-central-prd-ase1/ai-gateway/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/akamai-observability-mcp/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/alloy/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/argocd/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/aurva-dataplane/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/cert-manager/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/clickhouse/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/computeclass/contour-external-arm.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/computeclass/contour-external-cc.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-0-arm.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-0-cc.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-1-arm.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-1-cc.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-intra-0-arm.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-intra-0-cc.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-intra-1-arm.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-intra-1-cc.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/computeclass/contour-shared-arm.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/computeclass/contour-shared-cc.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/contour-ca-issuer/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/contour-external-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/contour-external/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/coredns/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/coroot-node-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/elasticsearch-mcp/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/etcd/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/fireworks-ai/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/fluentd-copy/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/kubectl-mcp-server/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/kyverno/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/kyverno/policies/protect-namespace.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/kyverno/policies/restrict-replicas.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/loadtester/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/opentelemetry-claude-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/opentelemetry-codex-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/opentelemetry-coralogix/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/temporal/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert-secured-stateful/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert-secured/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/victoriametrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/victoriametrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1/victoriametrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/contour-external-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/contour-external/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/coredns/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/ingress-nginx/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/opentelemetry-deployment/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-central-prd-ase1c/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/README.md delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/alloy/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/aurva-dataplane/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/cert-manager/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/contour-ca-issuer/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/contour-external/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/coredns/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/coroot-node-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/grafana/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/ingress-nginx-external/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/ingress-nginx-internal/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/ingress-nginx-secured/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/kubectl-mcp-server/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/kubernetes-dashboard/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/kyverno/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/kyverno/policies/protect-namespace.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/kyverno/policies/restrict-replicas.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/loadtester/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/opentelemetry-coralogix/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-agent-startree/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert-startree-stateful/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert-startree/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-insert-startree/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-select-startree/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-storage-startree/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoriametrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoriametrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1/victoriametrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/contour-external/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/coredns/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/ingress-nginx-external/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/ingress-nginx-internal/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/ingress-nginx/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-dataengg-prd-ase1c/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/README.md delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/alloy/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/aurva-dataplane/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/cert-manager/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-0-arm.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-1-arm.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-dataproc-arm.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-intra-0-arm.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-intra-1-arm.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-shared-arm.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/computeclass/g2-standard-16-l4-compute-class.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/computeclass/g2-standard-4-l4-compute-class.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/computeclass/g2-standard-8-l4-300gb-compute-class.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/computeclass/g2-standard-8-l4-compute-class.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/contour-ca-issuer/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/contour-internal-dataproc/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/coredns/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/coroot-node-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/ingress-nginx-internal/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/kubectl-mcp-server/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/kubernetes-dashboard/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/kyverno/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/kyverno/policies/protect-namespace.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/kyverno/policies/restrict-replicas.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/loadtester/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/opentelemetry-coralogix/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/paused-container/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert-secured-stateful/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert-secured/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/victoriametrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/victoriametrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1/victoriametrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/alloy/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/aurva-dataplane/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/coredns/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/ingress-nginx-internal/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/kubernetes-dashboard/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/loadtester/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/opentelemetry-coralogix/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/paused-container/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-alert/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/README.md delete mode 100644 helm-overrides/k8s-demand-prd-ase1/alloy/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/aurva-dataplane/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/cert-manager/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/contour-ca-issuer/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/contour-external/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/coredns/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/coroot-node-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/kubectl-mcp-server/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/kyverno/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/kyverno/policies/protect-namespace.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/kyverno/policies/restrict-replicas.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/loadtester/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/node-thp-config/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/opentelemetry-coralogix/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert-secured-stateful/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert-secured/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/victoriametrics-agent-secondary/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/victoriametrics-agent-shared/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/victoriametrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/victoriametrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1/victoriametrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/contour-external/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/coredns/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/ingress-nginx/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-demand-prd-ase1c/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/README.md delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/ingress-nginx-external/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/ingress-nginx-internal/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-alert/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/victoriametrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/victoriametrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-di-prd-ase1/victoriametrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-notebook-prd-ase1/README.md delete mode 100644 helm-overrides/k8s-dengspark-notebook-prd-ase1/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-notebook-prd-ase1/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-notebook-prd-ase1/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-notebook-prd-ase1/ingress-nginx-external/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-notebook-prd-ase1/ingress-nginx-internal/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-notebook-prd-ase1/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-notebook-prd-ase1/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-notebook-prd-ase1/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-notebook-prd-ase1/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-notebook-prd-ase1/rancher/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-alert/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/README.md delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/ingress-nginx-external/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/ingress-nginx-internal/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-alert/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/victoriametrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/victoriametrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-dengspark-prd-ase1/victoriametrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/README.md delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/ingress-nginx-internal/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-alert/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/victoriametrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/victoriametrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-dscispark-prd-ase1/victoriametrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/README.md delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/aurva-dataplane/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/cert-manager/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/computeclass/g2-standard-16-l4-compute-class.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/computeclass/g2-standard-4-l4-compute-class.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/computeclass/g2-standard-8-l4-300gb-compute-class.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/computeclass/g2-standard-8-l4-compute-class.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/contour-ca-issuer/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/coredns/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/external-dns-services/contour-internal-0-dsgpu-prd-envoy-headless-external-dns-svc.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/external-dns/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/ingress-nginx-internal/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/kubectl-mcp-server/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/loadtester/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/opentelemetry-coralogix/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/paused-container/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/victoriametrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/victoriametrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-dsgpu-prd-ase1/victoriametrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/README.md delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/alloy/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/aurva-dataplane/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/cert-manager/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/contour-external/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/coredns/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/kubectl-mcp-server/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/kyverno/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/kyverno/policies/protect-namespace.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/kyverno/policies/restrict-replicas.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/loadtester/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/opentelemetry-coralogix/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-alert/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/victoriametrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/victoriametrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-farmiso-prd-ase1/victoriametrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/README.md delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/cert-manager/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/coredns/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-alert/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-sec-admin-ase1/coredns/custom-values.yaml delete mode 100644 helm-overrides/k8s-sec-admin-ase1/deepfence-console/custom-values.yaml delete mode 100644 helm-overrides/k8s-sec-admin-ase1/deepfence-router/custom-values.yaml delete mode 100644 helm-overrides/k8s-sec-admin-ase1/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-sec-admin-ase1/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-sec-admin-ase1/ingress-nginx/custom-values.yaml delete mode 100644 helm-overrides/k8s-sec-admin-ase1/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-sec-admin-ase1/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/k8s-sec-admin-ase1/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-sec-admin-ase1/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-sec-admin-ase1/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-sec-admin-ase1/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/k8s-sec-admin-ase1/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-sec-admin-ase1/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-sec-admin-ase1/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-sec-admin-ase1/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/bifrost/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/cert-manager/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/compute-class/cc-contour-vmagent-od-compute-class.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/compute-class/cc-shared-cost-optimised-compute-class.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/compute-class/cc-shared-devops-od-compute-class.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/compute-class/cc-shared-gpu-a2-compute-class.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/compute-class/cc-shared-gpu-a3-compute-class.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/compute-class/cc-starrocks-n2-highmem-16.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/compute-class/cc-starrocks-n2-highmem-32.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/compute-class/cc-starrocks-n2-highmem-8.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/contour-ca-issuer/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/contour-external/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/coredns/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/deepgram-onprem/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/ingress-nginx-internal/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/kubectl-mcp-server/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/kyverno/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/kyverno/policies/protect-namespace.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/kyverno/policies/request-limit-policy.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/kyverno/policies/restrict-replicas.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-shared-int-ase1/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-dev-ase1/deepgram-onprem/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/README.md delete mode 100644 helm-overrides/k8s-supply-prd-ase1/alloy/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/aurva-dataplane/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/cert-manager/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/contour-ca-issuer/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/contour-external/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/coredns/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/coroot-node-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/deepgram-onprem-v2/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/deepgram-onprem/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/fluentd-sumoduolite-np/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/kubectl-mcp-server/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/kubernetes-dashboard/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/kyverno/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/kyverno/policies/protect-namespace.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/kyverno/policies/restrict-replicas.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/loadtester/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/node-thp-config/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/opentelemetry-coralogix/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/opentelemetry-daemonset-medium-np/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/telegraf-operator-custom/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert-secured-stateful/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert-secured/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoria-metrics-cluster-alert/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoria-metrics-cluster/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoria-metrics-select-rs-test1/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoria-metrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoriametrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoriametrics-insert-ht/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoriametrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoriametrics-select-ht/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoriametrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoriametrics-storage-ht/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1/victoriametrics-storage/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/conntrack-adjuster/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/contour-cert-checker/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/contour-external/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/contour-internal-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/contour-internal-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/contour-internal-intra-0/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/contour-internal-intra-1/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/coredns/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/external-secrets/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/flagger/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/fluentd/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/ingress-nginx/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/keda/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/kube-dns/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/kube-events/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/kube-state-metrics/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/opentelemetry-daemonset/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/prometheus-node-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/telegraf-operator/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/victoria-metrics-agent/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/victoria-metrics-insert/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/victoria-metrics-select/custom-values.yaml delete mode 100644 helm-overrides/k8s-supply-prd-ase1c/victoria-metrics-storage/custom-values.yaml delete mode 100644 index.md delete mode 100644 manifests/jenkins-filestore-caching/dev/pv.yaml delete mode 100644 manifests/jenkins-filestore-caching/dev/pvc.yaml delete mode 100644 manifests/jfrog-filestore-data/dev/pv.yaml delete mode 100644 manifests/jfrog-filestore-data/dev/pvc.yaml delete mode 100644 manifests/nginx-static-content/central/nginx-static-content-manifest.yaml delete mode 100644 manifests/nginx-static-content/demand/nginx-static-content-manifest.yaml delete mode 100644 manifests/nginx-static-content/farmiso/nginx-static-content-manifest.yaml delete mode 100644 manifests/nginx-static-content/supply/nginx-static-content-manifest.yaml delete mode 100644 manifests/priorityclass/gke-central-prd-ase1a/priorityclass-high.yaml delete mode 100644 manifests/priorityclass/gke-central-prd-ase1a/priorityclass-low.yaml delete mode 100644 manifests/priorityclass/gke-dataengg-prd-ase1a/priorityclass-high.yaml delete mode 100644 manifests/priorityclass/gke-dataengg-prd-ase1a/priorityclass-low.yaml delete mode 100644 manifests/priorityclass/gke-datascience-prd-ase1a/priorityclass-high.yaml delete mode 100644 manifests/priorityclass/gke-datascience-prd-ase1a/priorityclass-low.yaml delete mode 100644 manifests/priorityclass/gke-datascience-prd-ase1a/spot-termination-handler.yaml delete mode 100644 manifests/priorityclass/gke-dsgpu-prd-ase1a/priorityclass-high.yaml delete mode 100644 manifests/priorityclass/gke-dsgpu-prd-ase1a/priorityclass-low.yaml delete mode 100644 manifests/priorityclass/gke-farmiso-prd-ase1a/priorityclass-high.yaml delete mode 100644 manifests/priorityclass/k8s-central-prd-ase1/priorityclass-high.yaml delete mode 100644 manifests/priorityclass/k8s-central-prd-ase1/priorityclass-low.yaml delete mode 100644 manifests/priorityclass/k8s-dataengg-prd-ase1/priorityclass-high.yaml delete mode 100644 manifests/priorityclass/k8s-dataengg-prd-ase1/priorityclass-low.yaml delete mode 100644 manifests/priorityclass/k8s-datascience-prd-ase1/priorityclass-high.yaml delete mode 100644 manifests/priorityclass/k8s-datascience-prd-ase1/priorityclass-low.yaml delete mode 100644 manifests/priorityclass/k8s-demand-prd-ase1/priorityclass-high.yaml delete mode 100644 manifests/priorityclass/k8s-demand-prd-ase1/priorityclass-low.yaml delete mode 100644 manifests/priorityclass/k8s-farmiso-prd-ase1/priorityclass-high.yaml delete mode 100644 manifests/priorityclass/k8s-farmiso-prd-ase1/priorityclass-low.yaml delete mode 100644 manifests/priorityclass/k8s-sec-admin-ase1/priorityclass-high.yaml delete mode 100644 manifests/priorityclass/k8s-sec-admin-ase1/priorityclass-low.yaml delete mode 100644 manifests/priorityclass/k8s-supply-prd-ase1/priorityclass-high.yaml delete mode 100644 manifests/priorityclass/k8s-supply-prd-ase1/priorityclass-low.yaml delete mode 100644 manifests/storageclass/gke-central-prd-ase1a/hyperdisk-balanced.yaml delete mode 100644 manifests/storageclass/gke-central-prd-ase1a/pd-standard-retain.yaml delete mode 100644 manifests/storageclass/gke-central-prd-ase1a/pulse-nfs-sc-prd.yaml delete mode 100644 manifests/storageclass/gke-central-prd-ase1a/sc-pd-ssd.yaml delete mode 100644 manifests/storageclass/gke-central-prd-ase1a/sc-pd-standard.yaml delete mode 100644 manifests/storageclass/gke-datascience-prd-ase1a/enterprise-multishare-rwx.yaml delete mode 100644 manifests/storageclass/gke-datascience-prd-ase1a/enterprise-rwx.yaml delete mode 100644 manifests/storageclass/gke-datascience-prd-ase1a/hyperdisk-balanced.yaml delete mode 100644 manifests/storageclass/gke-datascience-prd-ase1a/pd-standard-retain.yaml delete mode 100644 manifests/storageclass/gke-datascience-prd-ase1a/premium-rwx.yaml delete mode 100644 manifests/storageclass/gke-datascience-prd-ase1a/pulse-nfs-sc-prd.yaml delete mode 100644 manifests/storageclass/gke-datascience-prd-ase1a/pulse-nfs-sc-secured-prd.yaml delete mode 100644 manifests/storageclass/gke-datascience-prd-ase1a/regional-rwx.yaml delete mode 100644 manifests/storageclass/gke-datascience-prd-ase1a/sc-filestore-standard.yaml delete mode 100644 manifests/storageclass/gke-datascience-prd-ase1a/standard-rwx-retain.yaml delete mode 100644 manifests/storageclass/gke-datascience-prd-ase1a/standard-rwx.yaml delete mode 100644 manifests/storageclass/gke-datascience-prd-ase1a/zonal-rwx.yaml delete mode 100644 post-commit-scripts/._commit-metric.sh delete mode 100644 post-commit-scripts/._runner.sh delete mode 100644 post-commit-scripts/commit-metric.sh delete mode 100644 post-commit-scripts/runner.sh delete mode 100644 pre-commit-scripts/._cac-validate.sh delete mode 100644 pre-commit-scripts/._runner.sh delete mode 100644 pre-commit-scripts/._trufflehog-hook.sh delete mode 100644 pre-commit-scripts/._yaakhook.sh delete mode 100644 pre-commit-scripts/cac-validate.sh delete mode 100644 pre-commit-scripts/runner.sh delete mode 100644 pre-commit-scripts/trufflehog-hook.sh delete mode 100644 pre-commit-scripts/yaakhook.sh delete mode 100644 repository.yaml delete mode 100644 skills/infra/add-infra-tool.md delete mode 100644 skills/infra/bump-chart-version.md delete mode 100644 skills/infra/check-cluster-health.md delete mode 100644 skills/infra/diagnose-deployment.md delete mode 100644 skills/infra/diagnose-scheduling.md delete mode 100644 skills/infra/onboard-app.md delete mode 100644 wiki/analyses/ADR-A1-cache-vs-upstream-charts.md delete mode 100644 wiki/analyses/ADR-A2-blue-green-sibling-pattern.md delete mode 100644 wiki/analyses/ADR-A3-per-cluster-scheduling.md delete mode 100644 wiki/analyses/ADR-A4-raw-manifest-sidecars-in-helm-overrides.md delete mode 100644 wiki/analyses/ADR-A5-manual-sync-default-for-infra.md delete mode 100644 wiki/entities/DevOps Infra Helm Charts.md diff --git a/CLAUDE.md b/CLAUDE.md deleted file mode 100644 index 66e0587..0000000 --- a/CLAUDE.md +++ /dev/null @@ -1,106 +0,0 @@ -# CLAUDE.md — `devops-infra-helm-charts` -# Auto-generated by /meesho-init. Edit freely — re-running suggests improvements, not overwrites. - -> Agent entry point for Meesho's infrastructure Helm values + cached/forked charts repo. -> -> **Repo role:** GitOps source-of-truth for *what* infrastructure tooling runs on Meesho's GKE fleet, *where*, and *with what values*. Sister repo `devops-infra-argo-config` is the routing layer — it holds the Argo CD `Application` / `ApplicationSet` manifests that point at paths in this repo. A merge to `main` is a deploy event: Argo CD on each cluster reconciles from `main`. -> -> **Layer:** **Layer 1 — Agent-Writable** (config repo). Generate diffs, open PRs, do **not** apply directly. The default safety property is reviewer discipline + the Argo CD Sync click on each cluster. The longer-term goal is tool-mediated edits via a `helm-values-tool`; until that exists, direct edits to `helm-overrides///custom-values.yaml` via PR are the supported path. Direct edits to `helm-templates//` are gated — see NEVER DO. -> -> **Out of scope:** application/service code (lives in service repos), Argo CD Application manifests (sister repo `devops-infra-argo-config`), workload-cluster `kubectl apply` operations (incident response, not authoring). -> -> See [docs/architecture.md](docs/architecture.md) for the full deploy lifecycle, cluster fleet, chart inventory, hook details, and gotchas. - -## NEVER DO - -- **NEVER** commit or push directly to `main`. Always work on a feature/fix branch and open a PR. A merge to `main` triggers Argo CD reconciliation against the live fleet. -- **NEVER** force push (`git push --force`). If absolutely required, use `--force-with-lease`. -- **NEVER** make requests to, curl, query, or interact with production endpoints: `int.meesho.int`, `prd.meesho.int`, `int.mrouter.int`, `prd.mrouter.int`, `*.meeshogcp.in`. These are production/pre-prod systems — any accidental call can affect live traffic or data. -- **NEVER** introduce backward-incompatible changes to chart `values.yaml` keys, image tags pinned in overrides, or `fullnameOverride` strings without explicit user approval. Argo CD will silently reconcile the change across every cluster that consumes the chart, and live releases (Service DNS, PVC binding) depend on the existing names. -- **NEVER** commit secrets in any form. The TruffleHog pre-commit hook is the last line of defense — **NEVER bypass it** with `--no-verify`, `git commit -n`, or by removing the hook. Real secrets belong in `external-secrets` (per-cluster) backed by GCP Secret Manager / Vault, not in `custom-values.yaml`. -- **NEVER** copy a `custom-values.yaml` from one cluster directory to another without rewriting `nodeSelector`, `tolerations`, and any `computeClass` references. Each cluster has a bespoke node-pool topology — see `contour-nodeselector-tolerations-summary.md`. GKE Autopilot clusters (`k8s-central-prd-ase1`, `k8s-dsgpu-prd-ase1`, `k8s-shared-int-ase1`) use `cloud.google.com/compute-class:` keys; standard clusters use `dedicated:` keys. Wrong values strand pods on wrong nodes or leave them pending. -- **NEVER** edit files under `helm-templates//templates/` or `values.yaml` casually. Most are vanilla upstream charts pulled via `helm pull`. Edits silently fork the chart and get clobbered on the next upstream sync. If a fork is intentional, document the reason in that chart's `README.md` and call it out in the PR. -- **NEVER** delete a `` / `-` sibling without confirming no Argo Application in `devops-infra-argo-config` still references it. Versioned siblings (`argo-cd-green`, `contour-v1.33.3`, `keda-2.17.1`, `opentelemetry-collector-latest`, `sonarqube-old`, `victoria-metrics-cluster-latest`, etc.) exist to support in-flight blue-green migrations — both versions may be live simultaneously. -- **NEVER** edit `manifests/storageclass/*.yaml` or `manifests/priorityclass//*.yaml` without a PR-level reviewer. These are cluster-wide singletons — a wrong StorageClass affects every PVC; a wrong PriorityClass changes scheduling priority for every pod that references it. -- **NEVER** bump a `Chart.yaml` `dependencies[].version` without (a) reading the upstream changelog for breaking template changes, (b) re-running `helm dependency update` to refresh `Chart.lock`, and (c) calling out the bump in the PR description. -- **NEVER** "normalize" values across clusters in the same PR as a feature change. Surgical edits only — touch the cluster × application that was asked, leave the rest. Cross-cluster cleanups belong in their own PR. -- **NEVER** change `fullnameOverride` values in any `custom-values.yaml`. They are load-bearing — Service DNS names, PVC bindings, ConfigMap references, and Argo Application names downstream depend on them being stable. -- **NEVER** treat Argo CD `Application` / `ApplicationSet` manifests as part of this repo. They live in the sister repo `github.com/Meesho/devops-infra-argo-config`. Changes to routing, sync policies, or Application paths are PRs against that repo, not this one. - -## Repo at a glance - -GitOps Helm values + cached/forked charts for Meesho's GKE infra fleet. Sister repo `devops-infra-argo-config` holds the Argo `Application` / `ApplicationSet` manifests that point at paths in this repo. **A merge to `main` is a deploy** — Argo CD on each cluster reconciles from `main`. - -> See [docs/architecture.md](docs/architecture.md) for the full deploy lifecycle, cluster fleet, chart inventory, hook details, and gotchas. - -## Repository layout - -| Path | Purpose | -|------|---------| -| `helm-templates//` | 74 cached/forked upstream charts (Argo CD, Contour, VictoriaMetrics, Grafana, Mimir, Loki, Tempo, Vault, Keda, Kyverno, Jenkins, JFrog, etc.). Some are thin wrappers (deps in `Chart.yaml`); some carry full vendored `templates/`. | -| `helm-overrides///custom-values.yaml` | Cluster × application Helm values overrides. Edited daily. | -| `helm-overrides///.yaml` | Raw manifests applied alongside the Helm release (e.g., `computeclass/*-cc.yaml`, `elastic-cluster/argo-launch.yaml`, `external-dns-services/*.yaml`). | -| `manifests/storageclass/`, `manifests/priorityclass//` | Cluster-wide singletons. High blast radius. | -| `manifests/{jenkins-filestore-caching,jenkins-gcs-caching,jfrog-filestore-data}/{dev,prd}/` | Per-env one-shot PV/PVC manifests. | -| `pre-commit-scripts/` | TruffleHog secret scan (active); CAC and Yaak hooks (no-op here, gated on paths this repo doesn't have). | -| `post-commit-scripts/` | Cursor AI commit metric collector (background, non-blocking). | -| `repository.yaml` | Owners (auto-managed). Primary: `siddharth.pal@meesho.com`. Secondary: `samarth.nag@meesho.com`. | -| `contour-nodeselector-tolerations-summary.md` | Per-cluster Contour scheduling matrix. **Read before editing any Contour values.** | - -## Cluster naming - -| Pattern | Meaning | -|---------|---------| -| `k8s--prd-ase1[c]` | Standard GKE prod cluster, BU-owned. BUs: `central`, `central-mqkafka`, `supply`, `supply-dev`, `demand`, `dataengg`, `datascience`, `dengspark`, `dengspark-di`, `dengspark-notebook`, `dscispark`, `dsgpu`, `farmiso`, `ml-platform`, `admin`, `sec-admin`, `devops-admin`. All in `asia-southeast1`, fleet `meesho-admin-prd-0622`. | -| `k8s-shared-int-ase1` | Shared **integration** (pre-prod) cluster. Only non-prod cluster in the repo. | -| `k8s-aurva-prd-ase1` | Aurva integration. Minimal override set. | -| `db--...` | Auto-named dataplane/data-tier clusters. Minimal overrides (`kube-state-metrics`, `victoria-metrics-agent`). Use `fullnameOverride: -dbc--prd`. | - -## Editing workflow - -1. Branch off `main`. Don't push to `main`. -2. Edit the **single** `helm-overrides///custom-values.yaml` (or `/.yaml`) the task targets. Don't drive-by-edit other apps in the same dir. -3. If the change is per-cluster, mirror **only** if the user asked — and rewrite per-cluster scheduling fields (see NEVER DO). -4. `git commit` — TruffleHog runs automatically. If it blocks, fix the secret (don't bypass). -5. Open a PR. Reviewer checks blast radius. Merge to `main`. -6. Argo CD on the target cluster syncs (auto or manual sync, per the Application's `syncPolicy` in `devops-infra-argo-config`). - -## Common patterns - -- **Multi-Contour clusters** — `contour-external`, `contour-external-1`, `contour-internal-0`, `contour-internal-1`, `contour-internal-intra-{0,1}` are all separate Helm releases per cluster. Each has its own node pool / dedicated taint or compute class. Cross-reference `contour-nodeselector-tolerations-summary.md`. -- **Versioned chart siblings** — `argo-cd` ↔ `argo-cd-green`, `contour` ↔ `contour-v1.33.3`, `keda` ↔ `keda-2.17.1`, `opentelemetry-collector` ↔ `-latest`, `victoria-metrics-{cluster,agent}` ↔ `-latest`, `sonarqube` ↔ `sonarqube-old`. The variant is the upgrade target. Both can be live at once. -- **Image registry** — production overrides pin `asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/` (Meesho's Artifact Registry mirror), not Docker Hub. -- **Secrets via External Secrets Operator** — most clusters have an `external-secrets/` override; secrets are sourced from GCP Secret Manager / Vault. Reference secret names; never paste secret values. - -## Quick reference - -This repo has no build, test, or lint commands. Everything is declarative YAML. Useful local commands: - -| Task | Command | -|------|---------| -| Install pre-commit hooks (one-time) | `pre-commit install --hook-type pre-commit --hook-type pre-push --hook-type post-commit` | -| Re-run pre-commit on staged changes | `pre-commit run` | -| Render a chart locally to inspect output | `helm template helm-templates/ -f helm-overrides///custom-values.yaml` | -| Refresh subchart deps after `Chart.yaml` bump | `helm dependency update helm-templates/` | -| Diff a release against the rendered template | `helm diff upgrade helm-templates/ -f helm-overrides///custom-values.yaml` (requires `helm-diff` plugin and kube context) | -| Lint a chart | `helm lint helm-templates/` | -| Find which clusters override a given app | `find helm-overrides -maxdepth 2 -type d -name ''` | - -## Layer constraint summary - -| Operation | Layer | Agent action | -|-----------|-------|--------------| -| Edit `helm-overrides///custom-values.yaml` (single cluster × app) | **Layer 1** | Generate the diff, open a PR. Reviewer + Argo CD Sync click are the safety gates. | -| Add a new app override under an existing cluster | **Layer 1** | Same — pair with the matching `Application` PR in `devops-infra-argo-config`. | -| Add a new cluster directory under `helm-overrides/` | **Layer 1 (HIGH RISK)** | Open a PR; pair with the cluster's `ApplicationSet` change in the sister repo. Verify per-cluster `nodeSelector` / `tolerations` / `computeClass` are written from scratch, not copied. | -| Bump a `Chart.yaml` `dependencies[].version` in `helm-templates//` | **Layer 1 (HIGH RISK)** | Read upstream changelog, run `helm dependency update`, refresh `Chart.lock`, call out the bump in the PR. | -| Edit `helm-templates//templates/` or `values.yaml` | **Layer 1 (HIGH RISK)** | Most charts are vanilla upstream; an edit silently forks the chart and gets clobbered on the next sync. Only allowed if the fork is intentional and documented in that chart's `README.md`. | -| Delete a versioned sibling chart (`-green`, `-vX.Y.Z`, `-latest`, `-old`) | **Layer 1 (HIGH RISK)** | Confirm no Argo Application in `devops-infra-argo-config` still references it. | -| Edit `manifests/storageclass/*.yaml` or `manifests/priorityclass//*.yaml` | **Layer 1 (HIGH RISK)** | Cluster-wide singleton; affects every PVC / scheduling priority. Requires platform-team review. | -| Run `helm install` / `helm upgrade` against a live cluster | **Out of scope** | This is GitOps; in-cluster mutation is incident response, not authoring. Use Argo CD UI Sync. | -| Run `kubectl apply -f` against a workload cluster | **Out of scope** | Drift will reappear on next reconciliation. | -| Hand-edit `repository.yaml` | **Layer 3** | Owned by `registry-bootstrap` automation. Refuse + redirect upstream. | -| Edit Argo `Application` / `ApplicationSet` manifests | **Out of scope (sister repo)** | These live in `github.com/Meesho/devops-infra-argo-config`. Open the PR there. | -| Recommend a curl/probe against `int.meesho.int`, `prd.meesho.int`, `*.mrouter.int`, or workload `*.meeshogcp.in` services | **Layer 3** | Production traffic surfaces — refuse. (Pre-commit hook telemetry to `observe.meeshogcp.in` is automated infrastructure, not agent-initiated.) | - - diff --git a/claude/00-overview.md b/claude/00-overview.md deleted file mode 100644 index 212a2be..0000000 --- a/claude/00-overview.md +++ /dev/null @@ -1,63 +0,0 @@ -> Per AI Blitz Plan §claude. Layer: 1. Repo: devops-infra-helm-charts. - -# 00 — Overview - -## Why this repo exists - -`devops-infra-helm-charts` is the GitOps source-of-truth for **what infrastructure tooling runs on Meesho's GKE fleet, where, and with what values**. It is one of two repos that together compose the platform's deploy plane: - -- **This repo** — values + cached/forked charts. Answers "what does cluster X's Argo CD agent stack look like?" -- **Sister repo** — [`Meesho/devops-infra-argo-config`](https://github.com/Meesho/devops-infra-argo-config) — Argo `Application` / `ApplicationSet` manifests. Answers "which cluster pulls which path from the values repo, with what sync policy?" - -A merge to `main` here is a **deploy event**: every Argo CD instance whose Application points at a touched path will reconcile, on its own cadence (auto-sync) or on a human Sync click (manual-sync — the prod default). - -See [`../wiki/entities/DevOps Infra Helm Charts.md`](../wiki/entities/DevOps%20Infra%20Helm%20Charts.md) for the conceptual model and [`../docs/architecture.md`](../docs/architecture.md) for the full deploy lifecycle. - -## What's in it - -| Top-level | Role | -|-----------|------| -| `helm-templates//` | 74 cached or forked upstream charts (Argo CD, Contour, VictoriaMetrics, Mimir, Loki, Tempo, Vault, Keda, Kyverno, Jenkins, JFrog, Grafana…). | -| `helm-overrides///custom-values.yaml` | Per-cluster × per-app Helm values. Edited daily. | -| `helm-overrides///.yaml` | Raw manifests applied alongside the Helm release (compute-class definitions, external-DNS records, etc.). | -| `manifests/storageclass/`, `manifests/priorityclass//` | Cluster-wide singletons. High blast radius. | -| `manifests/{jenkins,jfrog}-…/{dev,prd}/` | Per-env one-shot PV/PVC manifests. | -| `pre-commit-scripts/` | TruffleHog (active, blocking); CAC + Yaak (no-op here). | -| `post-commit-scripts/` | Cursor AI commit metric collector (background). | -| `repository.yaml` | Owners, auto-managed by `registry-bootstrap`. | -| `contour-nodeselector-tolerations-summary.md` | Per-cluster Contour scheduling matrix. | - -Detailed walkthrough in [`./01-repo-structure.md`](./01-repo-structure.md). - -## What's NOT in it - -- Argo CD `Application` / `ApplicationSet` manifests — those live in the **sister repo**. See [`../docs/global/coding-guidelines/argocd.md`](../docs/global/coding-guidelines/argocd.md). -- Application / service code — lives in service repos. -- Workload-cluster `kubectl apply` operations — that's incident response, not authoring. -- Production endpoint probes (`*.meesho.int`, `*.mrouter.int`, `*.meeshogcp.in`) — never call from an agent. See [`../docs/global/SANCTITY_RULES.md`](../docs/global/SANCTITY_RULES.md). - -## How a change reaches a cluster - -1. Branch off `main`. -2. Edit a single `helm-overrides///custom-values.yaml`. -3. Local dry-run: `helm template helm-templates/ -f helm-overrides///custom-values.yaml`. -4. Commit — TruffleHog runs (blocking). Never `--no-verify`. -5. Open a PR. Reviewer is the first safety gate. -6. Merge to `main`. -7. Argo CD on the target cluster either auto-reconciles (low-risk leaves) or waits for a human Sync click (prod infra default). - -That click is the **second** safety gate. The combined property (reviewer + Sync) is the system's safety floor while a tool-mediated edit path (`helm-values-tool`) is still being built. - -See [`./05-deploy-lifecycle.md`](./05-deploy-lifecycle.md) for the full flow with failure modes, and [`./08-pre-commit-and-hooks.md`](./08-pre-commit-and-hooks.md) for the hook details. - -## Layer classification - -This repo is **Layer 1 — Agent-Writable** (config repo). Most edits are agent-eligible via PR. Several operations are Layer 1 *high-risk* or Layer 3 (refuse). The full mapping is in the root `CLAUDE.md` *Layer constraint summary* table; the agent-facing summary is [`../docs/global/AGENT_BOUNDARIES.md`](../docs/global/AGENT_BOUNDARIES.md) and [`../docs/global/SANCTITY_RULES.md`](../docs/global/SANCTITY_RULES.md). - -## Where to go next - -- New to the repo: read [`./01-repo-structure.md`](./01-repo-structure.md) → [`./02-cluster-fleet.md`](./02-cluster-fleet.md) → [`./05-deploy-lifecycle.md`](./05-deploy-lifecycle.md). -- About to edit values: [`./04-override-hierarchy.md`](./04-override-hierarchy.md) and [`../docs/global/coding-guidelines/helm-values.md`](../docs/global/coding-guidelines/helm-values.md). -- About to bump a chart: [`../docs/platform/procedures/update-chart-version.md`](../docs/platform/procedures/update-chart-version.md). -- About to onboard an app: [`../skills/infra/onboard-app.md`](../skills/infra/onboard-app.md) and [`../docs/platform/procedures/onboard-app-to-cluster.md`](../docs/platform/procedures/onboard-app-to-cluster.md). -- Glossary: [`./10-glossary-and-references.md`](./10-glossary-and-references.md). diff --git a/claude/01-repo-structure.md b/claude/01-repo-structure.md deleted file mode 100644 index e71b786..0000000 --- a/claude/01-repo-structure.md +++ /dev/null @@ -1,70 +0,0 @@ -> Per AI Blitz Plan §claude. Layer: 1. Repo: devops-infra-helm-charts. - -# 01 — Repository structure - -Directory walkthrough with the *why* attached. The root `CLAUDE.md` *Repository layout* table is the canonical short version; this page extends it with rationale and links into the rest of the tree. - -## `helm-templates//` — cached / forked upstream charts - -About 74 chart directories. Three flavours: - -1. **Vanilla upstream cache** — pulled via `helm pull /` and committed verbatim. Edits to `templates/` here silently fork the chart and get clobbered on the next refresh. Most charts are this flavour. See [`../wiki/analyses/ADR-A1-cache-vs-upstream-charts.md`](../wiki/analyses/ADR-A1-cache-vs-upstream-charts.md). -2. **Thin wrapper** — `Chart.yaml` declares `dependencies:`, `templates/` is small or empty, and the real content lives in the subchart. Used to bind multiple sub-charts as one Argo Application. -3. **Intentional fork** — `templates/` is meaningfully edited. Each fork should explain itself in that chart's `README.md`. Forks are rare and need explicit owner approval. See [`../docs/platform/procedures/fork-upstream-chart.md`](../docs/platform/procedures/fork-upstream-chart.md). - -Why cache at all? Network/ingress robustness for Argo CD on every cluster, and a stable target for the values to bind against. Trade-off and alternatives in [`../wiki/analyses/ADR-A1-cache-vs-upstream-charts.md`](../wiki/analyses/ADR-A1-cache-vs-upstream-charts.md). - -## `helm-overrides///custom-values.yaml` - -The day-to-day editing surface. Path encodes the destination: - -- `` → directory name matches the GKE cluster name (`k8s-central-prd-ase1`, `k8s-shared-int-ase1`, `k8s-aurva-prd-ase1`, `db--…`). -- `` → directory name matches the Argo Application name in the sister repo (and usually matches the chart name in `helm-templates/`, but doesn't have to — many apps target a versioned-sibling chart). - -The content is a Helm values overlay merged onto `helm-templates//values.yaml` at render time. Schema: [`../docs/platform/schemas/custom-values-schema.md`](../docs/platform/schemas/custom-values-schema.md). Composition rules: [`./04-override-hierarchy.md`](./04-override-hierarchy.md). - -## `helm-overrides///.yaml` - -Raw Kubernetes manifests dropped alongside the Helm release. They are NOT consumed by Helm — Argo applies them directly. Common patterns: - -- `computeclass/*-cc.yaml` — GKE Autopilot `ComputeClass` objects. -- `elastic-cluster/argo-launch.yaml` — Elasticsearch Operator CR. -- `external-dns-services/*.yaml` — `Service` objects with `external-dns` annotations to publish DNS records. - -Schema and conventions: [`../docs/platform/schemas/raw-manifest-sidecar-schema.md`](../docs/platform/schemas/raw-manifest-sidecar-schema.md). Why they live here rather than in dedicated manifest dirs: [`../wiki/analyses/ADR-A4-raw-manifest-sidecars-in-helm-overrides.md`](../wiki/analyses/ADR-A4-raw-manifest-sidecars-in-helm-overrides.md). - -## `manifests/storageclass/` and `manifests/priorityclass//` - -Cluster-wide singletons. A wrong StorageClass affects every PVC; a wrong PriorityClass changes scheduling priority for every pod that references it. Two-reviewer policy. Schema: [`../docs/platform/schemas/storageclass-priorityclass-schema.md`](../docs/platform/schemas/storageclass-priorityclass-schema.md). Blast-radius detail: [`./07-singletons-and-blast-radius.md`](./07-singletons-and-blast-radius.md). - -## `manifests/{jenkins-filestore-caching,jenkins-gcs-caching,jfrog-filestore-data}/{dev,prd}/` - -Per-env one-shot PV / PVC manifests for stateful systems that pre-date a Helm-managed model. Treated as immutable once bound; resize via PVC `resources.requests.storage` rather than re-creating. - -## `pre-commit-scripts/` - -- **TruffleHog secret scan** — active, blocking. NEVER bypass. See [`./08-pre-commit-and-hooks.md`](./08-pre-commit-and-hooks.md) and [`../docs/global/SANCTITY_RULES.md`](../docs/global/SANCTITY_RULES.md). -- **CAC, Yaak hooks** — gated on file paths this repo doesn't have, so they no-op here. Same scripts run for real in service repos. - -## `post-commit-scripts/` - -- **Cursor AI commit metric collector** — background, non-blocking. Posts metric pings to `observe.meeshogcp.in`. This is platform-managed infrastructure, not agent-initiated. - -## `repository.yaml` - -Owners + secondary owners. Managed by the `registry-bootstrap` automation. Editing by hand is on the don't-touch list — see [`../docs/global/escalation-matrix.md`](../docs/global/escalation-matrix.md) row 4. - -## `contour-nodeselector-tolerations-summary.md` - -Per-cluster Contour scheduling matrix at the repo root. **Read this before any Contour values edit.** Multi-Contour pattern (`contour-external`, `contour-external-1`, `contour-internal-{0,1}`, `contour-internal-intra-{0,1}`) is detailed in [`./03-chart-inventory.md`](./03-chart-inventory.md). - -## `docs/`, `claude/`, `skills/`, `wiki/` - -The Blitz documentation tree. Entry points: - -- [`../docs/architecture.md`](../docs/architecture.md) — full deploy lifecycle and gotchas. -- [`../docs/global/`](../docs/global/) — agent boundaries, sanctity rules, escalation, coding guidelines. -- [`../docs/platform/`](../docs/platform/) — procedures, runbooks, schemas. -- [`../skills/infra/`](../skills/infra/) — task playbooks. -- [`../wiki/`](../wiki/) — entity model and ADRs. -- [`./00-overview.md`](./00-overview.md) — top of this `claude/` index. diff --git a/claude/02-cluster-fleet.md b/claude/02-cluster-fleet.md deleted file mode 100644 index fb6cb43..0000000 --- a/claude/02-cluster-fleet.md +++ /dev/null @@ -1,80 +0,0 @@ -> Per AI Blitz Plan §claude. Layer: 1. Repo: devops-infra-helm-charts. - -# 02 — Cluster fleet - -The repo's cluster directories under `helm-overrides//` are the canonical list of clusters this platform serves. Naming follows three patterns; each implies a different scheduling primitive set, which is why **schedule fields are never copy-pasted between clusters** (see [`../docs/global/SANCTITY_RULES.md`](../docs/global/SANCTITY_RULES.md)). - -## Naming taxonomy - -### `k8s--prd-ase1[c]` - -Standard GKE prod cluster, BU-owned. All in `asia-southeast1`, fleet `meesho-admin-prd-0622`. Trailing `c` denotes a secondary cluster for the same BU. - -Known BUs in this repo: - -- `central`, `central-mqkafka` -- `supply`, `supply-dev` -- `demand` -- `dataengg`, `datascience`, `dengspark`, `dengspark-di`, `dengspark-notebook`, `dscispark` -- `dsgpu` -- `farmiso` -- `ml-platform` -- `admin`, `sec-admin`, `devops-admin` - -A few of these are GKE **Autopilot** clusters (different scheduling primitives — see below): -- `k8s-central-prd-ase1` -- `k8s-dsgpu-prd-ase1` -- `k8s-shared-int-ase1` - -The rest are **standard** GKE. - -### `k8s-shared-int-ase1` - -Shared **integration** (pre-prod) cluster — the only non-prod cluster in the repo. Used for integration testing of platform changes before they hit any prod cluster. Autopilot. - -### `k8s-aurva-prd-ase1` - -Aurva integration (third-party security tooling). Minimal override set. - -### `db--...` - -Auto-named dataplane / data-tier clusters. Each carries a minimal override set, typically `kube-state-metrics` and `victoria-metrics-agent` only. The `fullnameOverride` convention here is `-dbc--prd` so that metrics labels stay legible across the data-tier fleet. - -## Autopilot vs standard scheduling - -This is the dominant reason scheduling fields cannot be copy-pasted across clusters. - -### Standard GKE clusters - -Use: -- `nodeSelector.dedicated: ` -- `tolerations[].key: dedicated` -- Node pools are explicitly provisioned per workload class. - -### Autopilot clusters - -Use: -- `nodeSelector."cloud.google.com/compute-class": ` -- `tolerations[].key: cloud.google.com/compute-class` (when applicable) -- A `ComputeClass` raw manifest is dropped at `helm-overrides///computeclass/*-cc.yaml` to declare the class. See [`../docs/platform/schemas/raw-manifest-sidecar-schema.md`](../docs/platform/schemas/raw-manifest-sidecar-schema.md). - -Mismatched fields → pods strand on the wrong nodes or stay `Pending`. Diagnosis flow: [`../docs/platform/runbooks/pod-pending-scheduling.md`](../docs/platform/runbooks/pod-pending-scheduling.md). Background: [`../wiki/analyses/ADR-A3-per-cluster-scheduling.md`](../wiki/analyses/ADR-A3-per-cluster-scheduling.md). - -## Multi-Contour scheduling - -Multi-Contour clusters run `contour-external`, `contour-external-1`, `contour-internal-0`, `contour-internal-1`, `contour-internal-intra-0`, `contour-internal-intra-1` — each a separate Helm release pinned to its own node pool / dedicated taint or compute class. The per-cluster matrix (which Contour goes where) is documented in the repo-root `contour-nodeselector-tolerations-summary.md`. **Always cross-reference that file before touching a Contour values override.** - -## Cluster onboarding - -Adding a new cluster directory is Layer-1 *high-risk*. The procedure (paired PR with the sister repo's `ApplicationSet`) is in [`../docs/platform/procedures/onboard-new-cluster.md`](../docs/platform/procedures/onboard-new-cluster.md). - -## Cluster deboarding - -Removing a cluster is rare and requires draining Argo Applications first. There is no in-repo procedure today; escalate to the primary owner — see [`../docs/global/escalation-matrix.md`](../docs/global/escalation-matrix.md). - -## See also - -- [`./03-chart-inventory.md`](./03-chart-inventory.md) — versioned siblings + multi-Contour pattern -- [`./04-override-hierarchy.md`](./04-override-hierarchy.md) — how cluster + chart compose -- [`./07-singletons-and-blast-radius.md`](./07-singletons-and-blast-radius.md) — `manifests/priorityclass//` -- [`../contour-nodeselector-tolerations-summary.md`](../contour-nodeselector-tolerations-summary.md) diff --git a/claude/03-chart-inventory.md b/claude/03-chart-inventory.md deleted file mode 100644 index f923579..0000000 --- a/claude/03-chart-inventory.md +++ /dev/null @@ -1,82 +0,0 @@ -> Per AI Blitz Plan §claude. Layer: 1. Repo: devops-infra-helm-charts. - -# 03 — Chart inventory - -`helm-templates/` carries about 74 chart directories. They fall into three categories by structure and one cross-cutting category by lifecycle. - -## By structure - -### 1. Vanilla upstream cache - -Most charts are pulled verbatim via `helm pull /` and committed. The `templates/` are *not* edited. Editing them silently forks the chart and the edits get clobbered the next time someone refreshes the cache. The pre-commit hooks do not catch this — only PR review does. - -If a fork is intentional, it must be documented in the chart's `README.md` and called out in the PR. See [`../docs/platform/procedures/fork-upstream-chart.md`](../docs/platform/procedures/fork-upstream-chart.md). - -Background: [`../wiki/analyses/ADR-A1-cache-vs-upstream-charts.md`](../wiki/analyses/ADR-A1-cache-vs-upstream-charts.md). - -### 2. Thin wrapper - -`Chart.yaml` declares `dependencies:` pointing at one or more upstream charts. The local `templates/` is empty or carries only a thin glue manifest. Used so a single Argo Application can install a stack (e.g., kube-prometheus-stack carries Prometheus + Alertmanager + Grafana + node-exporter + kube-state-metrics together). - -When bumping a wrapper, refresh `Chart.lock` with `helm dependency update helm-templates/`. See [`../docs/platform/procedures/update-chart-version.md`](../docs/platform/procedures/update-chart-version.md). - -### 3. Intentional fork - -A small number of charts have deliberate `templates/` edits — local CRD patches, label injection, removed sub-resources we don't want, etc. Each fork should be self-documenting in its `README.md`. If the rationale is missing, treat the fork as suspect and escalate per [`../docs/global/escalation-matrix.md`](../docs/global/escalation-matrix.md). - -## By lifecycle — versioned siblings - -A chart family often has two siblings live simultaneously to support blue-green migrations: - -| Stable | Migration target | Used for | -|--------|------------------|----------| -| `argo-cd` | `argo-cd-green` | Green-deploy of the Argo CD control plane itself | -| `contour` | `contour-v1.33.3` | Pinned-version migration of the ingress data plane | -| `keda` | `keda-2.17.1` | Autoscaler version cutover | -| `opentelemetry-collector` | `opentelemetry-collector-latest` | OTel collector cutover | -| `victoria-metrics-cluster` | `victoria-metrics-cluster-latest` | VM cluster cutover | -| `victoria-metrics-agent` | `victoria-metrics-agent-latest` | VM agent cutover | -| `sonarqube` | (was forward; `sonarqube-old` retained) | SonarQube major cutover | - -The `-green` / `-vX.Y.Z` / `-latest` / `-old` suffix names the **migration target** (or in `-old`'s case, the kept-around predecessor). Both can be live at once on different clusters or even on the same cluster (different Argo Applications). Deletion of a sibling requires confirming zero references in the sister repo. See [`../wiki/analyses/ADR-A2-blue-green-sibling-pattern.md`](../wiki/analyses/ADR-A2-blue-green-sibling-pattern.md) and [`../docs/platform/procedures/blue-green-chart-migration.md`](../docs/platform/procedures/blue-green-chart-migration.md). - -## Multi-Contour pattern - -Contour is unique in that a single cluster runs **multiple separate Contour Helm releases**, each pinned to its own node pool / compute class. The standard set: - -- `contour-external` — public-facing ingress, primary -- `contour-external-1` — public-facing ingress, secondary (capacity / blue-green) -- `contour-internal-0`, `contour-internal-1` — internal mesh ingress, redundant pair -- `contour-internal-intra-0`, `contour-internal-intra-1` — intra-VPC ingress, redundant pair - -Per-cluster matrix of which release goes on which node pool: repo-root `contour-nodeselector-tolerations-summary.md`. Each release has its own `helm-overrides///custom-values.yaml`. - -## Notable individual charts - -| Chart | Notes | -|-------|-------| -| `argo-cd` / `argo-cd-green` | Self-managing — Argo CD installs itself. Sync policy must be careful. | -| `vault` | Stateful HA on Raft. Edits to seal config or HA storage require platform-team review. | -| `external-secrets` | Source of truth for secret materialization on each cluster. See [`./06-secrets-and-identity.md`](./06-secrets-and-identity.md). | -| `kyverno` | Cluster-policy enforcement. Edits change admission behaviour for all workloads. | -| `kube-prometheus-stack` | Bundles Prometheus + Alertmanager. Alert rules pages on-call — validate PromQL. See [`../docs/global/coding-guidelines/observability.md`](../docs/global/coding-guidelines/observability.md). | -| `external-dns` | Publishes DNS records to Cloud DNS. Often paired with sidecar `external-dns-services/*.yaml` raw manifests. | -| `cert-manager` | Issues TLS certs (Let's Encrypt + Vault). | - -## Image registry convention - -Production overrides pin images to Meesho's Artifact Registry mirror: - -``` -asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/ -``` - -…rather than Docker Hub directly. Mirroring decouples deploys from upstream registry availability and rate limits. - -## See also - -- [`./01-repo-structure.md`](./01-repo-structure.md) -- [`./02-cluster-fleet.md`](./02-cluster-fleet.md) -- [`./04-override-hierarchy.md`](./04-override-hierarchy.md) -- [`../docs/platform/procedures/update-chart-version.md`](../docs/platform/procedures/update-chart-version.md) -- [`../docs/platform/procedures/fork-upstream-chart.md`](../docs/platform/procedures/fork-upstream-chart.md) diff --git a/claude/04-override-hierarchy.md b/claude/04-override-hierarchy.md deleted file mode 100644 index 45a626d..0000000 --- a/claude/04-override-hierarchy.md +++ /dev/null @@ -1,72 +0,0 @@ -> Per AI Blitz Plan §claude. Layer: 1. Repo: devops-infra-helm-charts. - -# 04 — Override hierarchy - -How the per-cluster override file, the chart's default values, and any raw-manifest sidecars compose at deploy time. - -## The three layers - -For a given Argo Application targeting `` × ``, the rendered manifests come from three sources: - -1. **Chart defaults** — `helm-templates//values.yaml`. The upstream values, possibly customized in an intentional fork. This is the lowest precedence. -2. **Cluster override** — `helm-overrides///custom-values.yaml`. Helm-merged on top of (1). This is what the agent edits day-to-day. -3. **Raw manifest sidecars** — `helm-overrides///.yaml` (and optionally subdirectories like `computeclass/`, `external-dns-services/`). These are NOT consumed by Helm. Argo applies them directly to the cluster, in the same Application. - -The Argo Application in the sister repo declares which `path:` (the override directory) and which `helm.valueFiles:` to use. Conventionally the Application points at the override directory and lists `custom-values.yaml`; raw sidecars in the same directory are picked up by Argo's manifest discovery. - -## Helm merge semantics - -Helm performs a **deep merge** of (2) over (1): - -- Maps merge key-by-key. -- Lists are **replaced wholesale**, not merged. This is the most common surprise — to extend an upstream list (`tolerations`, `extraArgs`, `extraEnv`), copy the upstream list into the override and edit there. Don't write a list expecting it to append. -- `null` in the override deletes the key set in defaults. - -If you need surgical list editing rather than wholesale replacement, you must fork the chart and rewrite the template — almost never the right call. See [`../docs/platform/procedures/fork-upstream-chart.md`](../docs/platform/procedures/fork-upstream-chart.md). - -## Dry-running the merge - -Always render before pushing. The command from the root `CLAUDE.md` Quick reference table: - -``` -helm template helm-templates/ \ - -f helm-overrides///custom-values.yaml -``` - -For wrapper charts (`Chart.yaml` `dependencies:`), refresh subcharts first: - -``` -helm dependency update helm-templates/ -``` - -For raw sidecars, validate separately: - -``` -kubectl apply --dry-run=client -f helm-overrides///.yaml -``` - -## Where each piece of config belongs - -| Config | Goes in | Why | -|--------|---------|-----| -| Image tag pin (production registry) | `custom-values.yaml` | Per-cluster pinning is the whole point of the override layer. | -| `replicaCount`, resource requests | `custom-values.yaml` | Per-cluster capacity tuning. | -| `nodeSelector`, `tolerations`, `computeClass` | `custom-values.yaml` | Per-cluster node-pool topology. **Never copy-paste across clusters.** | -| `fullnameOverride` | `custom-values.yaml` | Pinned to keep Service DNS / PVC binding stable. **Never change** an existing one. | -| Helm-managed Service / Deployment / ConfigMap | chart's `templates/` (don't touch) | Owned by upstream chart. | -| External-DNS record bound to a Service the chart doesn't manage | `external-dns-services/*.yaml` raw sidecar | Not part of the chart's surface. | -| `ComputeClass` definition (Autopilot) | `computeclass/*-cc.yaml` raw sidecar | Cluster-scoped object the chart can't render. | -| StorageClass / PriorityClass | `manifests/storageclass/`, `manifests/priorityclass//` | Cluster-wide singleton, separate from any one Application. | -| Secret values | **External Secrets Operator** + GCP Secret Manager / Vault | Never in `custom-values.yaml`. See [`./06-secrets-and-identity.md`](./06-secrets-and-identity.md). | - -## Schema details - -- Override schema: [`../docs/platform/schemas/custom-values-schema.md`](../docs/platform/schemas/custom-values-schema.md) -- Raw-sidecar schema: [`../docs/platform/schemas/raw-manifest-sidecar-schema.md`](../docs/platform/schemas/raw-manifest-sidecar-schema.md) -- Singleton schema: [`../docs/platform/schemas/storageclass-priorityclass-schema.md`](../docs/platform/schemas/storageclass-priorityclass-schema.md) - -## See also - -- [`./00-overview.md`](./00-overview.md) -- [`./05-deploy-lifecycle.md`](./05-deploy-lifecycle.md) -- [`../docs/global/coding-guidelines/helm-values.md`](../docs/global/coding-guidelines/helm-values.md) diff --git a/claude/05-deploy-lifecycle.md b/claude/05-deploy-lifecycle.md deleted file mode 100644 index 7834222..0000000 --- a/claude/05-deploy-lifecycle.md +++ /dev/null @@ -1,103 +0,0 @@ -> Per AI Blitz Plan §claude. Layer: 1. Repo: devops-infra-helm-charts. - -# 05 — Deploy lifecycle - -End-to-end story of how a values change reaches a live cluster, where the safety gates are, and what happens when they fail. - -## The path - -``` -edit override (branch) → commit (TruffleHog runs) → push → PR - → reviewer approves → merge to main - → Argo CD on cluster reconciles (auto-sync OR human Sync click) - → manifests applied → workload changes -``` - -## Stage 1 — Branch and edit - -- Branch off `main`. Never push to `main` directly. -- Edit one `helm-overrides///custom-values.yaml` (or its raw sidecars). -- No drive-by edits, no cross-cluster normalization in the same PR. See [`../docs/global/SANCTITY_RULES.md`](../docs/global/SANCTITY_RULES.md). - -## Stage 2 — Local validation - -- `helm template` against the override — confirms render succeeds. -- For wrappers: `helm dependency update` first. -- For raw sidecars: `kubectl apply --dry-run=client`. - -If render fails locally, it will fail in Argo CD's `OutOfSync → SyncFailed`. Fix before pushing. - -## Stage 3 — Commit - -`git commit` triggers pre-commit hooks: - -- **TruffleHog** — blocking. Real secrets bounce. Never `--no-verify`. -- **CAC, Yaak** — gated on paths this repo doesn't have, no-op. - -Post-commit: - -- **Cursor metric collector** — background, non-blocking. Pings `observe.meeshogcp.in` with commit telemetry. Failure here does not block. - -Detail: [`./08-pre-commit-and-hooks.md`](./08-pre-commit-and-hooks.md). - -## Stage 4 — PR + review (safety gate 1) - -The reviewer's job: - -1. Confirm the change touches only the cluster × app named in the PR. -2. Confirm any per-cluster scheduling fields were rewritten, not copy-pasted. -3. Confirm `fullnameOverride` is unchanged. -4. Confirm no secret materializes in the file. -5. Confirm chart `Chart.yaml` dep bumps came with `Chart.lock` refresh and a changelog reference. -6. Confirm versioned-sibling deletes have no sister-repo references. - -If a chart fork is suspected, escalate per [`../docs/global/escalation-matrix.md`](../docs/global/escalation-matrix.md) row 1. - -## Stage 5 — Merge to `main` - -Merging is the deploy event. Argo CD on every cluster whose Application points at the changed path will move to `OutOfSync`. - -## Stage 6 — Argo CD reconcile (safety gate 2) - -Two reconciliation modes, set per Application in the sister repo: - -- **Manual sync** (prod default for infra) — Argo waits for a human Sync click. Engineer reviews the diff in the Argo UI before applying. -- **Auto-sync** — Argo applies on its own. Reserved for low-risk leaves (`kube-state-metrics`, monitoring agents). - -Background on the manual-sync default: [`../wiki/analyses/ADR-A5-manual-sync-default-for-infra.md`](../wiki/analyses/ADR-A5-manual-sync-default-for-infra.md). - -The Argo Application also defines: - -- `syncPolicy.automated.prune` — whether Argo deletes objects no longer in Git. -- `syncPolicy.automated.selfHeal` — whether Argo reverts manual cluster edits. -- `syncOptions` — `CreateNamespace`, `ServerSideApply`, `RespectIgnoreDifferences`, retry/backoff. -- Sync waves via annotations (in chart templates or sidecars). - -These all live in the **sister repo**, not here. See [`../docs/global/coding-guidelines/argocd.md`](../docs/global/coding-guidelines/argocd.md). - -## Failure modes - -| Failure | Where it surfaces | Read | -|---------|------------------|------| -| Render error in `helm template` | Argo Application status `ComparisonError` | Re-render locally; fix values | -| `OutOfSync → SyncFailed` after Sync click | Argo UI events | [`../docs/platform/runbooks/argocd-sync-failure.md`](../docs/platform/runbooks/argocd-sync-failure.md) | -| Pods land but stay `Pending` | `kubectl get pods` on target cluster | [`../docs/platform/runbooks/pod-pending-scheduling.md`](../docs/platform/runbooks/pod-pending-scheduling.md) | -| Ingress 5xx after Contour change | `contour-external` envoy logs / synthetic probes | [`../docs/platform/runbooks/ingress-down.md`](../docs/platform/runbooks/ingress-down.md) | -| Drift reappears after `kubectl edit` | `selfHeal: true` doing its job | Edit Git, not the cluster | - -## Sister-repo coupling - -Almost every non-trivial change is a **paired PR**: - -- New app on cluster: PR here (override) + PR in sister repo (Application). -- New cluster: PR here (cluster directory) + PR in sister repo (`ApplicationSet` cluster generator). -- Blue-green sibling cutover: PR here (sibling values) + PR in sister repo (Application `targetRevision` / chart path). - -Procedures: [`../docs/platform/procedures/onboard-app-to-cluster.md`](../docs/platform/procedures/onboard-app-to-cluster.md), [`../docs/platform/procedures/onboard-new-cluster.md`](../docs/platform/procedures/onboard-new-cluster.md), [`../docs/platform/procedures/blue-green-chart-migration.md`](../docs/platform/procedures/blue-green-chart-migration.md), [`../docs/platform/procedures/deboard-app.md`](../docs/platform/procedures/deboard-app.md). - -## See also - -- [`./00-overview.md`](./00-overview.md) -- [`./04-override-hierarchy.md`](./04-override-hierarchy.md) -- [`../docs/architecture.md`](../docs/architecture.md) -- [`../docs/global/agent-operations-guide.md`](../docs/global/agent-operations-guide.md) diff --git a/claude/06-secrets-and-identity.md b/claude/06-secrets-and-identity.md deleted file mode 100644 index 1b679e8..0000000 --- a/claude/06-secrets-and-identity.md +++ /dev/null @@ -1,70 +0,0 @@ -> Per AI Blitz Plan §claude. Layer: 1. Repo: devops-infra-helm-charts. - -# 06 — Secrets and identity - -This repo is **values + cached charts**, all of it world-readable from Git. No secret value should ever exist in a file here. The platform pattern is to express secrets as *references* and let the in-cluster machinery materialize them. - -## The components - -### External Secrets Operator (ESO) - -Each cluster carries an `external-secrets/` override directory. ESO runs in the cluster, reads `ExternalSecret` CRs, fetches the secret value from a backend (GCP Secret Manager or Vault), and materializes a Kubernetes `Secret` for workloads to mount. - -- The `ExternalSecret` CR refers to a backend by **name only** (e.g., `gcpsm/prod//api-token`). The CR is checked into Git; the value is not. -- The backend is configured per-cluster in `external-secrets/custom-values.yaml`. - -### GCP Secret Manager - -The default backend for most prod clusters. Secrets are project-scoped under the cluster's GCP project. Workload Identity binds the ESO service account to a Google service account that has `secretmanager.secretAccessor` on the relevant secrets. - -### Vault - -Used where additional capabilities are required — dynamic credentials, transit encryption, PKI. Vault runs in-cluster on Raft HA. Edits to Vault overrides (`helm-overrides//vault/custom-values.yaml`) — especially seal config, HA storage, autounseal — require platform-team review. Vault HA write-path failure is a paging incident. See [`../docs/global/escalation-matrix.md`](../docs/global/escalation-matrix.md) row 8. - -### Workload Identity - -GKE Workload Identity binds Kubernetes service accounts to Google service accounts via the `iam.gke.io/gcp-service-account` annotation. This is how pods authenticate to GCP APIs (Secret Manager, Cloud Storage, Pub/Sub) without long-lived keys. - -The annotation is set in the chart's values (`serviceAccount.annotations`) — agent-editable. The IAM binding itself is set up out-of-band via Terraform in the platform IaC repo, not here. - -## What never goes in this repo - -- Plain-text passwords, API tokens, certificates, private keys. -- Base64-encoded secrets in `Secret` manifests. -- TLS keys / certs (use `cert-manager` issuers + ESO references instead). -- GCP service-account JSON keys (Workload Identity replaces them). -- OAuth client secrets (Secret Manager → ESO). -- Webhook URLs that contain a credential token in the path. - -If the value would be useful to an attacker who clones this repo, it does not belong here. - -## TruffleHog — last line of defense - -The pre-commit hook scans staged content for high-entropy strings and known secret formats. **NEVER bypass.** - -- `git commit --no-verify` is blocked by Sanctity rule. -- If the hook flags a real secret, rotate the credential first (the moment it touched a Git working tree it is already half-burned), then move it to ESO. -- If the hook flags a false positive, fix the regex in `pre-commit-scripts/` rather than excluding the file. - -Detail: [`./08-pre-commit-and-hooks.md`](./08-pre-commit-and-hooks.md). - -## The pre-commit hook telemetry exception - -The post-commit Cursor metric collector POSTs to `observe.meeshogcp.in`. This is platform-managed automation, not agent-initiated, and is the only outbound call to a `*.meeshogcp.in` host the agent will ever observe in this repo. The agent must still refuse any *new* call to such hosts. See [`../docs/global/SANCTITY_RULES.md`](../docs/global/SANCTITY_RULES.md). - -## Common patterns - -| Pattern | Looks like | -|---------|-----------| -| Pod reads a Secret Manager value | `ExternalSecret` CR → ESO materializes `Secret` → pod mounts via `envFrom.secretRef` or `volumes.secret` | -| Pod calls a GCP API | KSA annotated with `iam.gke.io/gcp-service-account: @.iam.gserviceaccount.com` (Workload Identity) | -| TLS for an Ingress | `cert-manager` `Certificate` CR + Vault PKI or Let's Encrypt issuer | -| Vault dynamic DB credential | Vault DB secrets engine + ESO `VaultDynamicSecret` (or app-side Vault Agent sidecar) | - -## See also - -- [`./08-pre-commit-and-hooks.md`](./08-pre-commit-and-hooks.md) -- [`./07-singletons-and-blast-radius.md`](./07-singletons-and-blast-radius.md) -- [`../docs/global/SANCTITY_RULES.md`](../docs/global/SANCTITY_RULES.md) -- [`../docs/global/escalation-matrix.md`](../docs/global/escalation-matrix.md) -- [`../docs/global/coding-guidelines/helm-values.md`](../docs/global/coding-guidelines/helm-values.md) diff --git a/claude/07-singletons-and-blast-radius.md b/claude/07-singletons-and-blast-radius.md deleted file mode 100644 index e702095..0000000 --- a/claude/07-singletons-and-blast-radius.md +++ /dev/null @@ -1,85 +0,0 @@ -> Per AI Blitz Plan §claude. Layer: 1. Repo: devops-infra-helm-charts. - -# 07 — Singletons and blast radius - -Most files in this repo affect one cluster × one application. A small number of files affect **everything** on a cluster (or every cluster). These are the singletons. They get a different review bar. - -## `manifests/storageclass/*.yaml` - -Cluster-wide `StorageClass` objects. Every PVC on the cluster either references one of these by name or relies on the default annotation (`storageclass.kubernetes.io/is-default-class: "true"`). - -Wrong here means: - -- New PVCs bind to a different disk type (cost, latency, IOPS change). -- `volumeBindingMode` change (Immediate ↔ WaitForFirstConsumer) changes scheduling semantics for every stateful workload. -- Default-class flip changes behaviour of every chart that doesn't pin a class explicitly. - -**Layer-1 high risk.** Two reviewers, one of whom must be a cluster BU owner. Schema: [`../docs/platform/schemas/storageclass-priorityclass-schema.md`](../docs/platform/schemas/storageclass-priorityclass-schema.md). - -## `manifests/priorityclass//*.yaml` - -Cluster-wide `PriorityClass` objects, partitioned by cluster directory. Every pod that sets `spec.priorityClassName: ` resolves against this set. - -Wrong here means: - -- A `value:` change can swap which workloads preempt others under capacity pressure. -- A `globalDefault: true` flip changes behaviour of every pod that omits `priorityClassName`. -- Removing a `PriorityClass` referenced by a live workload causes admission failure on next pod create. - -**Layer-1 high risk.** Same review policy as StorageClass. Schema: same file as above. - -## `repository.yaml` - -Owners, secondary owners, repo metadata. Owned by the **`registry-bootstrap` automation**, not by humans. Hand edits will be reverted on the next `registry-bootstrap` run. - -**Layer-3 — refuse.** If asked to edit, redirect to `registry-bootstrap`. See [`../docs/global/escalation-matrix.md`](../docs/global/escalation-matrix.md) row 4. - -## `manifests/{jenkins-filestore-caching,jenkins-gcs-caching,jfrog-filestore-data}/{dev,prd}/` - -Per-env, one-shot PV / PVC manifests for stateful systems (Jenkins build cache, JFrog binary store). Bound to GCP Filestore or GCS. Once a PVC is bound to a PV with a real backend, you cannot move it without data migration. - -**Layer-1 high risk.** Resize via `resources.requests.storage` only; do not recreate. - -## Versioned-sibling chart deletion - -Deleting a chart directory under `helm-templates/` (e.g., removing `argo-cd-green` after a successful migration) is irreversible from Argo CD's point of view — any cluster whose Application still points at it will fail to render. - -**Layer-1 high risk.** Confirm zero references in [`Meesho/devops-infra-argo-config`](https://github.com/Meesho/devops-infra-argo-config) first. See [`../wiki/analyses/ADR-A2-blue-green-sibling-pattern.md`](../wiki/analyses/ADR-A2-blue-green-sibling-pattern.md) and [`../docs/global/escalation-matrix.md`](../docs/global/escalation-matrix.md) row 10. - -## Chart `templates/` edits - -Editing `helm-templates//templates/` or `values.yaml` of a vanilla-pulled chart silently forks it. The next refresh clobbers the edit, but until then it ships to every cluster that consumes the chart. - -**Layer-1 high risk.** Allowed only if the fork is intentional and documented in the chart's `README.md`. See [`../docs/platform/procedures/fork-upstream-chart.md`](../docs/platform/procedures/fork-upstream-chart.md). - -## Kyverno cluster policies - -`helm-overrides//kyverno/custom-values.yaml` configures admission policies. A new `enforce`-mode `ClusterPolicy` can block every pod admission on a cluster. - -**Layer-1 high risk.** Roll out in `audit` mode first, observe `PolicyReport`s, then flip to `enforce`. - -## Argo CD itself (`argo-cd` / `argo-cd-green`) - -Argo CD self-manages — it deploys itself from this repo. A bad values change can break the control plane that would otherwise heal it. Recovery requires `kubectl` access to apply a hand-rendered manifest. - -**Layer-1 high risk.** Always cut a green sibling first; never edit the live release directly. - -## Decision summary - -| Singleton | Layer | Review policy | -|-----------|-------|---------------| -| `manifests/storageclass/*.yaml` | 1 high-risk | Two reviewers, one cluster BU owner | -| `manifests/priorityclass//*.yaml` | 1 high-risk | Two reviewers, one cluster BU owner | -| `repository.yaml` | 3 | Refuse; redirect to `registry-bootstrap` | -| `manifests/{jenkins,jfrog}-…/{dev,prd}/` | 1 high-risk | Two reviewers; resize-only edits | -| Versioned-sibling chart deletion | 1 high-risk | Confirm sister-repo zero references | -| `helm-templates//templates/` edits | 1 high-risk | README must document the fork | -| Kyverno enforce-mode policy | 1 high-risk | Audit-mode rollout first | -| Argo CD self-managed values | 1 high-risk | Cut a green sibling first | - -## See also - -- [`./01-repo-structure.md`](./01-repo-structure.md) -- [`../docs/platform/schemas/storageclass-priorityclass-schema.md`](../docs/platform/schemas/storageclass-priorityclass-schema.md) -- [`../docs/global/SANCTITY_RULES.md`](../docs/global/SANCTITY_RULES.md) -- [`../docs/global/escalation-matrix.md`](../docs/global/escalation-matrix.md) diff --git a/claude/08-pre-commit-and-hooks.md b/claude/08-pre-commit-and-hooks.md deleted file mode 100644 index b8765bf..0000000 --- a/claude/08-pre-commit-and-hooks.md +++ /dev/null @@ -1,69 +0,0 @@ -> Per AI Blitz Plan §claude. Layer: 1. Repo: devops-infra-helm-charts. - -# 08 — Pre-commit and hooks - -What runs when you `git commit` here, what blocks, what doesn't, and why nothing should be bypassed. - -## Hook installation - -One-time, per clone: - -``` -pre-commit install \ - --hook-type pre-commit \ - --hook-type pre-push \ - --hook-type post-commit -``` - -If the hooks aren't installed, the local commit will skip them — but PR review is the catch-net, and a missed scan in a feature branch can still catch the secret before merge. - -## Active hooks - -### TruffleHog (pre-commit, blocking) - -Scans the staged content for high-entropy strings and known secret patterns (AWS keys, GCP service-account JSON, GitHub tokens, generic JWTs, etc.). - -- **Blocks the commit** on any positive match. -- **NEVER bypass** with `git commit --no-verify` or `git commit -n`. This is on the don't-touch list — see [`../docs/global/SANCTITY_RULES.md`](../docs/global/SANCTITY_RULES.md). -- If the hook fires on a **real secret**: stop, rotate the credential immediately (any value that touched a Git working tree is half-burned), then move to External Secrets Operator. See [`./06-secrets-and-identity.md`](./06-secrets-and-identity.md). -- If the hook fires on a **false positive**: fix the regex in `pre-commit-scripts/` rather than skip-listing the file. The fix is reusable across the org. - -### CAC and Yaak (pre-commit / pre-push, gated) - -These hooks exist in the platform's standard `.pre-commit-config.yaml`, but they are gated on file paths this repo doesn't carry (CAC config files, Yaak collections). They no-op here. The same scripts run for real in service repos. - -If a future change ever introduces matching paths, the hooks will start firing — read their messages and fix forward. Do not disable. - -## Background hooks - -### Cursor AI commit metric collector (post-commit, non-blocking) - -Posts a metric ping to `observe.meeshogcp.in` describing the commit (author, files touched, AI tool used). Runs in the background, does not block, and silently drops on failure. - -- This is **the only sanctioned outbound call to a `*.meeshogcp.in` host** the agent should ever observe in this repo. Agents must still refuse to *initiate* any such call themselves. See [`../docs/global/SANCTITY_RULES.md`](../docs/global/SANCTITY_RULES.md). -- If the post-commit script is failing, that's a platform issue — escalate per [`../docs/global/escalation-matrix.md`](../docs/global/escalation-matrix.md). Do not remove the script. - -## Why never `--no-verify` - -A bypassed pre-commit hook is invisible to the PR reviewer. Real secrets ship through merged PRs are very expensive to recover from: - -- The credential itself must be rotated everywhere it's used. -- The Git history must be force-rewritten (and even then, the Git push may be cached on a mirror). -- Any system that ingested the secret value (CI logs, Slack quotes, downstream forks) is now compromised. - -The 5 seconds saved bypassing the hook is a 5-day-or-more incident later. - -## When the hook is wrong - -Two kinds of false-positive: - -1. **Pattern over-matches** — TruffleHog regex matches a non-secret high-entropy string (a hash, a UUID, a build label). Fix: tighten the regex in `pre-commit-scripts/`. -2. **Genuine fixture / test data** — a fake-looking string in a chart's example values or test fixture. Fix: same — tighten the pattern, or move the fixture to a path TruffleHog already excludes (chart `templates/` test fixtures usually qualify). - -Either way, the fix is in the hook, not in the bypass. - -## See also - -- [`./06-secrets-and-identity.md`](./06-secrets-and-identity.md) -- [`../docs/global/SANCTITY_RULES.md`](../docs/global/SANCTITY_RULES.md) -- [`../docs/global/agent-operations-guide.md`](../docs/global/agent-operations-guide.md) diff --git a/claude/09-common-tasks.md b/claude/09-common-tasks.md deleted file mode 100644 index 77829fb..0000000 --- a/claude/09-common-tasks.md +++ /dev/null @@ -1,67 +0,0 @@ -> Per AI Blitz Plan §claude. Layer: 1. Repo: devops-infra-helm-charts. - -# 09 — Common tasks - -Index of the procedures and skills checked into the repo. Pick the task, follow the link, run the playbook. - -## Procedures (deeper, multi-step, often paired with sister-repo PR) - -| Task | Read | -|------|------| -| Onboard a new app to an existing cluster | [`../docs/platform/procedures/onboard-app-to-cluster.md`](../docs/platform/procedures/onboard-app-to-cluster.md) | -| Onboard a new cluster | [`../docs/platform/procedures/onboard-new-cluster.md`](../docs/platform/procedures/onboard-new-cluster.md) | -| Deboard / remove an app from a cluster | [`../docs/platform/procedures/deboard-app.md`](../docs/platform/procedures/deboard-app.md) | -| Bump a chart version (Chart.yaml deps) | [`../docs/platform/procedures/update-chart-version.md`](../docs/platform/procedures/update-chart-version.md) | -| Cut a blue-green sibling and migrate to it | [`../docs/platform/procedures/blue-green-chart-migration.md`](../docs/platform/procedures/blue-green-chart-migration.md) | -| Intentionally fork an upstream chart | [`../docs/platform/procedures/fork-upstream-chart.md`](../docs/platform/procedures/fork-upstream-chart.md) | - -## Skills (focused, single-task playbooks) - -| Skill | Read | -|-------|------| -| Bump a chart version (concise checklist) | [`../skills/infra/bump-chart-version.md`](../skills/infra/bump-chart-version.md) | -| Diagnose pods stuck `Pending` (scheduling) | [`../skills/infra/diagnose-scheduling.md`](../skills/infra/diagnose-scheduling.md) | -| Onboard an app (concise checklist) | [`../skills/infra/onboard-app.md`](../skills/infra/onboard-app.md) | - -## Runbooks (failure response) - -| Symptom | Read | -|---------|------| -| Argo CD `OutOfSync → SyncFailed` | [`../docs/platform/runbooks/argocd-sync-failure.md`](../docs/platform/runbooks/argocd-sync-failure.md) | -| Pods stuck `Pending` | [`../docs/platform/runbooks/pod-pending-scheduling.md`](../docs/platform/runbooks/pod-pending-scheduling.md) | -| Ingress 5xx after a Contour change | [`../docs/platform/runbooks/ingress-down.md`](../docs/platform/runbooks/ingress-down.md) | - -## Schemas (reference while editing) - -| Surface | Read | -|---------|------| -| `helm-overrides///custom-values.yaml` | [`../docs/platform/schemas/custom-values-schema.md`](../docs/platform/schemas/custom-values-schema.md) | -| Raw `.yaml` sidecars in override dirs | [`../docs/platform/schemas/raw-manifest-sidecar-schema.md`](../docs/platform/schemas/raw-manifest-sidecar-schema.md) | -| Cluster-wide `StorageClass` / `PriorityClass` | [`../docs/platform/schemas/storageclass-priorityclass-schema.md`](../docs/platform/schemas/storageclass-priorityclass-schema.md) | - -## Coding guidelines - -| Domain | Read | -|--------|------| -| Helm values conventions | [`../docs/global/coding-guidelines/helm-values.md`](../docs/global/coding-guidelines/helm-values.md) | -| Argo CD interaction model | [`../docs/global/coding-guidelines/argocd.md`](../docs/global/coding-guidelines/argocd.md) | -| Observability stack | [`../docs/global/coding-guidelines/observability.md`](../docs/global/coding-guidelines/observability.md) | - -## Operating discipline - -| Topic | Read | -|-------|------| -| Pre-flight + authoring loop | [`../docs/global/agent-operations-guide.md`](../docs/global/agent-operations-guide.md) | -| Don't-touch list | [`../docs/global/SANCTITY_RULES.md`](../docs/global/SANCTITY_RULES.md) | -| Layer classification | [`../docs/global/AGENT_BOUNDARIES.md`](../docs/global/AGENT_BOUNDARIES.md) | -| Escalation table | [`../docs/global/escalation-matrix.md`](../docs/global/escalation-matrix.md) | - -## ADRs (rationale, when you want to know *why*) - -| ADR | Read | -|-----|------| -| Why cache charts here vs pull from upstream at deploy | [`../wiki/analyses/ADR-A1-cache-vs-upstream-charts.md`](../wiki/analyses/ADR-A1-cache-vs-upstream-charts.md) | -| Why versioned-sibling charts (`-green`, `-vX.Y.Z`, `-latest`) | [`../wiki/analyses/ADR-A2-blue-green-sibling-pattern.md`](../wiki/analyses/ADR-A2-blue-green-sibling-pattern.md) | -| Why per-cluster scheduling fields cannot be shared | [`../wiki/analyses/ADR-A3-per-cluster-scheduling.md`](../wiki/analyses/ADR-A3-per-cluster-scheduling.md) | -| Why raw manifest sidecars live next to overrides | [`../wiki/analyses/ADR-A4-raw-manifest-sidecars-in-helm-overrides.md`](../wiki/analyses/ADR-A4-raw-manifest-sidecars-in-helm-overrides.md) | -| Why manual Argo sync is the prod default | [`../wiki/analyses/ADR-A5-manual-sync-default-for-infra.md`](../wiki/analyses/ADR-A5-manual-sync-default-for-infra.md) | diff --git a/claude/10-glossary-and-references.md b/claude/10-glossary-and-references.md deleted file mode 100644 index 9af4e4f..0000000 --- a/claude/10-glossary-and-references.md +++ /dev/null @@ -1,61 +0,0 @@ -> Per AI Blitz Plan §claude. Layer: 1. Repo: devops-infra-helm-charts. - -# 10 — Glossary and references - -Short definitions for the terms that recur in this repo's docs, and outbound links for deep dives. - -## Glossary - -**Argo CD** — GitOps continuous-delivery controller. Reconciles a cluster's actual state to a Git-declared desired state. Each cluster runs its own Argo CD instance; each Argo CD instance hosts a set of `Application` objects. - -**Application (Argo)** — A single deployable unit. Points at a Git repo + path + revision + chart-and-values config, and a destination (cluster + namespace). In our setup, the source path is in **this** repo; the Application manifest itself is in the **sister repo**. - -**ApplicationSet** — A controller-side template that fans out one Application per cluster (or per cluster × app). Used in the sister repo to express "deploy `victoria-metrics-agent` to every prod cluster" once instead of N times. - -**BU (Business Unit)** — Meesho-internal grouping that owns a cluster. Encoded in the cluster name: `k8s--prd-ase1[c]`. Examples: `central`, `supply`, `demand`, `dataengg`, `ml-platform`. - -**Autopilot (GKE)** — Google's managed-node-pool flavour of GKE. Scheduling primitives are different from standard GKE — uses `cloud.google.com/compute-class` instead of `dedicated:` taints. The repo has three Autopilot clusters: `k8s-central-prd-ase1`, `k8s-dsgpu-prd-ase1`, `k8s-shared-int-ase1`. See [`./02-cluster-fleet.md`](./02-cluster-fleet.md). - -**ESO (External Secrets Operator)** — In-cluster operator that reads `ExternalSecret` CRs and materializes Kubernetes `Secret` objects from a remote backend (GCP Secret Manager, Vault). The mechanism that keeps secret values out of this repo. See [`./06-secrets-and-identity.md`](./06-secrets-and-identity.md). - -**ComputeClass** — A GKE Autopilot CR that describes a node-pool selection policy (machine family, accelerators, spot eligibility). Workloads target a ComputeClass via `nodeSelector."cloud.google.com/compute-class": `. CRs live as raw sidecars in `helm-overrides///computeclass/`. - -**fullnameOverride** — A Helm values key consumed by most charts to fix the resource name prefix. **Load-bearing** — Service DNS names, PVC bindings, ConfigMap references all key off it. Never change for a live release. See [`../docs/global/SANCTITY_RULES.md`](../docs/global/SANCTITY_RULES.md). - -**Sister repo** — [`Meesho/devops-infra-argo-config`](https://github.com/Meesho/devops-infra-argo-config). Owns the Argo `Application` / `ApplicationSet` manifests that point at paths in this repo. See [`../docs/global/coding-guidelines/argocd.md`](../docs/global/coding-guidelines/argocd.md). - -**External Secrets** — short for the External Secrets Operator (above), or the `ExternalSecret` CR it consumes. - -**Blue-green sibling** — A second chart directory under `helm-templates/` (`-green`, `-vX.Y.Z`, `-latest`, `-old`) that exists alongside the stable chart to support a phased migration. Both can be live simultaneously. See [`../wiki/analyses/ADR-A2-blue-green-sibling-pattern.md`](../wiki/analyses/ADR-A2-blue-green-sibling-pattern.md). - -**helm-overrides** — Top-level dir holding per-cluster × per-app values overlays (`//custom-values.yaml`) and raw-manifest sidecars. The agent-edited surface. - -**helm-templates** — Top-level dir holding cached / forked upstream charts. Mostly read-only. Edits silently fork unless intentional. - -**Manual sync** — Argo `syncPolicy.automated` is unset; reconciliation requires a human Sync click in the Argo UI. The default for prod infra. See [`../wiki/analyses/ADR-A5-manual-sync-default-for-infra.md`](../wiki/analyses/ADR-A5-manual-sync-default-for-infra.md). - -**Workload Identity** — GKE feature that binds a Kubernetes service account to a Google service account via the `iam.gke.io/gcp-service-account` annotation. Replaces long-lived JSON service-account keys. - -**TruffleHog** — Pre-commit secret scanner. Active and blocking on this repo. Never bypass. - -**Layer 1 / Layer 3** — Agent authority classification from the AI Blitz Plan. Layer 1 = agent-writable (this repo, mostly). Layer 3 = refuse and redirect (`repository.yaml` edits, production endpoint probes). See [`../docs/global/AGENT_BOUNDARIES.md`](../docs/global/AGENT_BOUNDARIES.md). - -## References — internal - -- Repo-root `CLAUDE.md` — authoritative facts list. -- [`../docs/architecture.md`](../docs/architecture.md) — full deploy lifecycle and gotchas. -- [`../wiki/entities/DevOps Infra Helm Charts.md`](../wiki/entities/DevOps%20Infra%20Helm%20Charts.md) — entity model. -- [`./09-common-tasks.md`](./09-common-tasks.md) — task index. - -## References — external - -- **Sister repo** (Argo Application manifests): [`github.com/Meesho/devops-infra-argo-config`](https://github.com/Meesho/devops-infra-argo-config) -- **AI Blitz Plan** — internal Confluence; ask the primary owner for the current link. -- **Argo CD** — [argo-cd.readthedocs.io](https://argo-cd.readthedocs.io/), chart at [github.com/argoproj/argo-helm](https://github.com/argoproj/argo-helm/tree/main/charts/argo-cd) -- **Contour** — [projectcontour.io](https://projectcontour.io/), chart at [github.com/bitnami/charts/tree/main/bitnami/contour](https://github.com/bitnami/charts/tree/main/bitnami/contour) -- **VictoriaMetrics** — [docs.victoriametrics.com](https://docs.victoriametrics.com/), charts at [github.com/VictoriaMetrics/helm-charts](https://github.com/VictoriaMetrics/helm-charts) -- **HashiCorp Vault** — [developer.hashicorp.com/vault](https://developer.hashicorp.com/vault), chart at [github.com/hashicorp/vault-helm](https://github.com/hashicorp/vault-helm) -- **KEDA** — [keda.sh](https://keda.sh/), chart at [github.com/kedacore/charts](https://github.com/kedacore/charts) -- **Kyverno** — [kyverno.io](https://kyverno.io/), chart at [github.com/kyverno/kyverno/tree/main/charts](https://github.com/kyverno/kyverno/tree/main/charts) -- **External Secrets Operator** — [external-secrets.io](https://external-secrets.io/) -- **GKE Autopilot ComputeClass** — [cloud.google.com/kubernetes-engine/docs/concepts/autopilot-compute-classes](https://cloud.google.com/kubernetes-engine/docs/concepts/autopilot-compute-classes) diff --git a/contour-nodeselector-tolerations-summary.md b/contour-nodeselector-tolerations-summary.md deleted file mode 100644 index e76a365..0000000 --- a/contour-nodeselector-tolerations-summary.md +++ /dev/null @@ -1,70 +0,0 @@ -# Contour NodeSelector and Tolerations Summary - -## Summary by Cluster - -### k8s-central-prd-ase1 -| Contour | NodeSelector | Tolerations | -|---------|--------------|-------------| -| contour-external | `cloud.google.com/compute-class: contour-external-cc` | `cloud.google.com/compute-class: contour-external-cc, contour-shared-cc` | -| contour-external-1 | `dedicated: contour-external-1` | `dedicated: contour-external-1` | -| contour-internal-0 | `cloud.google.com/compute-class: contour-internal-0-cc` | `cloud.google.com/compute-class: contour-internal-0-cc, contour-shared-cc` | -| contour-internal-1 | `cloud.google.com/compute-class: contour-internal-1-cc` | `cloud.google.com/compute-class: contour-internal-1-cc, contour-shared-cc` | -| contour-internal-intra-0 | `cloud.google.com/compute-class: contour-intra-0-cc` | `cloud.google.com/compute-class: contour-intra-0-cc, contour-shared-cc` | -| contour-internal-intra-1 | `cloud.google.com/compute-class: contour-intra-1-cc` | `cloud.google.com/compute-class: contour-intra-1-cc, contour-shared-cc` | - - -### k8s-dataengg-prd-ase1 -| Contour | NodeSelector | Tolerations | -|---------|--------------|-------------| -| contour-external | `dedicated: contour-external` | `dedicated: contour-external` | -| contour-internal-0 | `dedicated: contour-internal-0` | `dedicated: contour-internal-0` | -| contour-internal-1 | `dedicated: contour-internal-1` | `dedicated: contour-internal-1, contour-shared` | -| contour-internal-intra-0 | `dedicated: contour-internal-0` | `dedicated: contour-internal-0` | -| contour-internal-intra-1 | `dedicated: contour-intra-1` | `dedicated: contour-intra-1, contour-shared` | - - -### k8s-datascience-prd-ase1 -| Contour | NodeSelector | Tolerations | -|---------|--------------|-------------| -| contour-internal-0 | `dedicated: contour-internal-0` | `dedicated: contour-internal-0` | -| contour-internal-1 | `dedicated: contour-internal-1-c4d` | `dedicated: contour-internal-1-c4d` | -| contour-internal-dataproc | `dedicated: contour-internal-1-c4d` | `dedicated: contour-internal-1-c4d` | -| contour-internal-intra-0 | `dedicated: contour-internal-0-c4d` | `dedicated: contour-internal-0-c4d` | -| contour-internal-intra-1 | `dedicated: contour-intra-1` | `dedicated: contour-intra-1, contour-shared` | - - -### k8s-demand-prd-ase1 -| Contour | NodeSelector | Tolerations | -|---------|--------------|-------------| -| contour-external | `dedicated: contour-external` | `dedicated: contour-external` | -| contour-internal-0 | `dedicated: contour-internal-0` | `dedicated: contour-internal-0, contour-shared` | -| contour-internal-1 | `dedicated: contour-internal-1` | `dedicated: contour-internal-0, contour-internal-1` | -| contour-internal-intra-0 | `dedicated: contour-intra-0` | `dedicated: contour-internal-0, contour-intra-0` | -| contour-internal-intra-1 | `dedicated: contour-intra-1` | `dedicated: contour-internal-0, contour-intra-1` | - - - -### k8s-farmiso-prd-ase1 -| Contour | NodeSelector | Tolerations | -|---------|--------------|-------------| -| contour-external | `dedicated: contour-external` | `dedicated: contour-external` | -| contour-internal-0 | `dedicated: contour-internal-0` | `dedicated: contour-internal-0` | -| contour-internal-intra-0 | `dedicated: contour-internal-0` | `dedicated: contour-internal-0` | - - -### k8s-supply-prd-ase1 -| Contour | NodeSelector | Tolerations | -|---------|--------------|-------------| -| contour-external | `dedicated: contour-external` | `dedicated: contour-external, contour-shared` | -| contour-internal-0 | `dedicated: contour-internal-0` | `dedicated: contour-internal-0, contour-shared` | -| contour-internal-1 | `dedicated: contour-internal-1` | `dedicated: contour-internal-1, contour-intra-0` | -| contour-internal-intra-0 | `dedicated: contour-intra-0` | `dedicated: contour-intra-0, contour-shared` | -| contour-internal-intra-1 | `dedicated: contour-intra-1` | `dedicated: contour-intra-1, contour-shared` | - -## Key Patterns Observed - -1. **Most clusters** use `dedicated` key for taints/tolerations with values matching the contour instance name -2. **GKE Autopilot clusters** (k8s-central-prd-ase1, k8s-dsgpu-prd-ase1, k8s-shared-int-ase1) use `cloud.google.com/compute-class` -3. **cert-checker** components use `-devops` toleration values -4. **Some clusters** have multiple tolerations for migration/shared scheduling (e.g., `contour-shared`, `contour-internal-1-new`) -5. **k8s-aurva-prd-ase1** is the only cluster without any nodeSelector/tolerations for contour diff --git a/docs/architecture.md b/docs/architecture.md deleted file mode 100644 index 31f750c..0000000 --- a/docs/architecture.md +++ /dev/null @@ -1,148 +0,0 @@ -# Architecture - -This is a **GitOps Helm values repository**, not a service repo. There is no application code, no build, no tests — only declarative YAML (Helm charts, value overrides, Kubernetes manifests) and two git-hook shell scripts. Argo CD is the runtime; merging to `main` is the deployment. - -## Section 1 — High-level design - -### Repo purpose - -Centralizes (1) cached/forked upstream Helm charts and (2) per-cluster value overrides for every infrastructure tool Meesho runs on its GKE fleet — observability (VictoriaMetrics, Mimir, Loki, Tempo, Grafana, OpenTelemetry, Pyroscope), ingress/edge (Contour/Envoy, ingress-nginx, cert-manager, external-dns, external-secrets), platform (Argo CD, Vault, Keda, Kyverno, Flagger, Jenkins, JFrog, Rancher, SonarQube), and data/AI (ClickHouse, Temporal, Superset, Deepgram, Aurva, Deepfence). It was carved out of `gcp-devops-admin` and is the source of truth for **what gets installed where, with what values**. - -### System context - -| Edge | Plays | -|------|-------| -| Sister repo `devops-infra-argo-config` | `github.com/Meesho/devops-infra-argo-config`. Holds the Argo CD `Application` / `ApplicationSet` manifests that point at this repo's `helm-overrides///` paths. Argo Application changes are PRs against that repo, not this one. | -| Argo CD instance(s) | Reconciles cluster state from this repo + the argo-config repo. One Argo CD per cluster (or per BU); each cluster has an `argocd/custom-values.yaml` here that configures *its own* Argo CD. | -| GKE cluster fleet | All consumers. Standard GKE clusters (`k8s--prd-ase1`) plus auto-named dataplane clusters (`db--...`). Region: `asia-southeast1`. Project fleet: `meesho-admin-prd-0622`. | -| TruffleHog webhook | `https://observe.meeshogcp.in/api/webhook` — pre-commit hook reports verified secret findings. | -| CAC API | `https://observe.meeshogcp.in/api/cac/repos` — pre-commit allowlist; this repo isn't on the list, so the `cac validate` hook is a no-op. | -| Cursor metrics API | `https://cursor-server.meeshogcp.in/api/v1/...` — post-commit hook ships per-commit Cursor AI usage metrics. Local-only side effect. | - -### Module boundaries - -| Top-level dir | What it owns | -|---------------|--------------| -| `helm-templates//` | Cached or forked upstream Helm chart. Touch only when a deliberate fork update is needed. Most are vanilla upstream — `Chart.yaml` + `templates/` + `values.yaml` (default upstream values). | -| `helm-templates/-/` | Versioned/blue-green sibling charts: `argo-cd` + `argo-cd-green`, `contour` + `contour-v1.33.3`, `keda` + `keda-2.17.1`, `opentelemetry-collector` + `-latest`, `victoria-metrics-cluster` + `-latest`, `victoria-metrics-agent` + `-latest`, `sonarqube` + `sonarqube-old`. The variant is the **target** of an in-flight chart upgrade — old version stays until the migration finishes. | -| `helm-overrides///custom-values.yaml` | Cluster × application override values. Argo CD's `helm.valueFiles` points here; values merge over the chart's own `values.yaml` (or the upstream subchart's defaults when the local chart is a thin wrapper). | -| `helm-overrides///.yaml` | Non-`custom-values` files: extra Kubernetes resources rendered by an Argo Application's `path:` (e.g., `computeclass/*-cc.yaml`, `elastic-cluster/argo-launch.yaml`, `mimir-distributed/alertmanager_config.yaml`, `external-dns-services/*.yaml`). These are not Helm values — they are raw manifests applied alongside the Helm release. | -| `manifests/` | One-shot, cluster-scoped resources applied outside the Helm flow: `storageclass/`, `priorityclass//`, `jenkins-filestore-caching/{dev,prd}/`, `jenkins-gcs-caching/`, `jfrog-filestore-data/{dev,prd}/`. These are singletons — wrong values affect every workload in the cluster. | -| `pre-commit-scripts/` | `runner.sh` (parallel exec), `trufflehog-hook.sh` (verified-secret scan + webhook), `cac-validate.sh` (no-op here — gated on `configs/` path), `yaakhook.sh` (no-op here — gated on `api-collections/` path). | -| `post-commit-scripts/` | `runner.sh` + `commit-metric.sh` — Cursor AI commit metric collector (forks to background; never blocks). | -| `repository.yaml` | Owners (auto-managed by registry-bootstrap). | - -### Architecture philosophy - -**Reliability-first, surgical edits.** A bad values change can take down ingress, observability, or an entire cluster. Two rules govern every change: -1. **Reliability-first** — production blast radius is huge; mirror the existing pattern of neighbor cluster overrides; never delete keys without checking what depends on them; preserve explicit limits and HPA bounds. -2. **Surgical** — touch only what was asked. Don't refactor surrounding values, don't "normalize" across clusters in the same PR, don't drive-by-edit other charts in the same cluster's directory. - -### Data flow (deployment lifecycle) - -``` -edit helm-overrides///custom-values.yaml - │ - ▼ -git commit ──► pre-commit hooks (TruffleHog secrets scan) - │ - ▼ -git push ──► PR → review → merge to main - │ - ▼ -Argo CD on each cluster polls this repo + devops-infra-argo-config - │ - ▼ -Application sync: helm template -f → apply to cluster - │ - ▼ -post-commit hook ships Cursor AI metrics (background, non-blocking) -``` - -Argo CD resolves `` from `helm-templates/` (when the Application uses `repoURL` of this repo with `path: helm-templates/`) or pulls upstream by reading the local `Chart.yaml` dependencies (e.g., `argo-cd/Chart.yaml` declares `argo-cd 7.7.23` from `argoproj.github.io/argo-helm`). - -### Cross-cutting concerns - -- **Secret management** — TruffleHog pre-commit hook (`pre-commit-scripts/trufflehog-hook.sh`) blocks any verified secret. Reports go to the security webhook with content hash + commit + file/line metadata. NEVER bypass. Real secrets are externalised via `external-secrets` (per-cluster override exists in most clusters) backed by GCP Secret Manager / Vault. -- **Per-cluster scheduling** — every cluster has its own `nodeSelector` / `tolerations` / `computeclass` topology. The `contour-nodeselector-tolerations-summary.md` at the repo root documents the current matrix. GKE Autopilot clusters (`k8s-central-prd-ase1`, `k8s-dsgpu-prd-ase1`, `k8s-shared-int-ase1`) use `cloud.google.com/compute-class` keys; standard clusters use `dedicated:` keys. Copying values between clusters without rewriting these is a reliable way to schedule pods on the wrong nodes. -- **Versioned chart migrations** — when upgrading a chart, the new version lives as a sibling dir (`argo-cd-green`, `contour-v1.33.3`, `keda-2.17.1`) until cutover. Both directories may be referenced by Argo Applications during the transition. Don't delete the old sibling without confirming no Application still points at it. -- **Image registry** — most overrides pin images to `asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/` (Meesho's internal Artifact Registry mirror), not upstream Docker Hub. - -## Section 2 — Low-level details - -### Cluster fleet (helm-overrides/) - -Each top-level dir under `helm-overrides/` is one cluster. Two naming conventions: - -| Convention | Example | Owner / type | -|------------|---------|--------------| -| `k8s--prd-ase1` (and `-prd-ase1c`) | `k8s-central-prd-ase1`, `k8s-supply-prd-ase1`, `k8s-dataengg-prd-ase1` | Standard GKE cluster, BU-owned (central, supply, demand, dataengg, datascience, dengspark, dscispark, dsgpu, farmiso, ml-platform, admin, sec-admin, devops-admin, central-mqkafka). Project varies per BU (e.g., `meesho-supply-prd`, `meesho-datascience-prd`). All in `asia-southeast1`. | -| `k8s-shared-int-ase1` | (only) | Shared **integration** (pre-prod) cluster. The only non-prod cluster in this repo. | -| `k8s-aurva-prd-ase1` | (only) | Aurva integration. Limited override set (contour-internal, rancher only). | -| `db--...` | `db-2516183257845181-c-1204-195038-428` | Auto-named dataplane / data-tier clusters. Override sets are minimal (typically `kube-state-metrics` + `victoria-metrics-agent` only). `fullnameOverride` values use `dbc--prd` form (e.g., `dbc-dsci-prd`). | -| `k8s-supply-dev-ase1` | (only) | Dev/sandbox supply cluster. | - -Per-cluster READMEs (where present) say: "This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster." - -### Application directory layout (per cluster) - -Inside `helm-overrides//`, each subdirectory is an Argo CD Application. Common shapes: - -| Layout | Meaning | Example | -|--------|---------|---------| -| `/custom-values.yaml` | Single Helm release; values merged onto a chart from `helm-templates/` | `argocd/custom-values.yaml`, `etcd/custom-values.yaml`, `contour-internal-0/custom-values.yaml` | -| `/.yaml` (no custom-values.yaml) | Argo Application's `path:` points here; raw manifests applied | `computeclass/contour-external-cc.yaml`, `elastic-cluster/argo-launch.yaml`, `mimir-distributed/alertmanager_config.yaml`, `external-dns-services/*.yaml` | -| Both | Helm release + sidecar raw manifests | rare; check the matching Application in `devops-infra-argo-config` | - -Multiple Contour instances per cluster is the norm — `contour-external`, `contour-external-1`, `contour-internal-0`, `contour-internal-1`, `contour-internal-intra-0`, `contour-internal-intra-1`. Each maps to a different node pool / dedicated node taint or compute class. The mapping per cluster is recorded in `contour-nodeselector-tolerations-summary.md` at the repo root — read before adding/changing a Contour instance. - -### Helm chart inventory - -| Category | Charts in `helm-templates/` | -|----------|----------------------------| -| Argo / GitOps | `argo-cd`, `argo-cd-green` | -| Ingress / edge | `contour`, `contour-v1.33.3`, `contour-ca-issuer`, `contour-cert-checker`, `ingress-nginx`, `cert-manager`, `external-dns`, `external-secrets` | -| Observability — metrics | `prometheus-node-exporter`, `prometheus-stackdriver-exporter`, `kube-state-metrics`, `kube-events`, `victoria-metrics-{single,cluster,cluster-latest,agent,agent-latest,alert,alert-stateful,alerts-config,auth,mcp}`, `vm-alert-config`, `mimir-distributed`, `pmm`, `telegraf-operator` | -| Observability — logs/traces/profiles | `fluentd`, `loki-distributed`, `tempo-distributed`, `pyroscope`, `alloy`, `opentelemetry-collector`, `opentelemetry-collector-latest`, `opentelemetry-operator`, `elastalert2`, `coroot-node-agent`, `deepfence-console`, `deepfence-router` | -| UI / dashboards | `grafana`, `grafana-edge`, `grafana-mcp`, `kubernetes-dashboard`, `superset`, `uptime-kuma` | -| Workflow / CI/CD | `jenkins`, `jfrog`, `sonarqube`, `sonarqube-old`, `flagger`, `keda`, `keda-2.17.1`, `kyverno`, `loadtester`, `temporal`, `dind`, `canary-bot-gcp`, `paused-container` | -| Networking / DNS | `coredns`, `kube-dns`, `bifrost`, `conntrack-adjuster`, `node-thp-config` | -| Data / search / DB | `clickhouse`, `etcd`, `vault`, `elasticsearch-mcp`, `eck-operator`, `athens-proxy` | -| AI / 3rd-party | `aurva-dataplane`, `deepgram-onprem`, `rancher` | - -74 charts total. Each has its own `Chart.yaml`. 19 also have a `Chart.lock` (charts with subchart dependencies that have been resolved with `helm dependency update`). Many of the local "charts" (e.g., `argo-cd/Chart.yaml`) are thin wrappers that declare the upstream chart as a dependency in `Chart.yaml` — the actual templates come from upstream. Others (e.g., `contour/`) carry a full vendored `templates/` tree. - -### Manifests (singletons) - -| Path | Scope | What it is | -|------|-------|------------| -| `manifests/storageclass/*.yaml` | Cluster-wide | StorageClasses: `pd-standard-retain-dr`, `sc-filestore-standard`, `sc-pd-ssd`, `sc-pd-standard`. Wrong change affects every PVC. | -| `manifests/priorityclass//*.yaml` | Per-cluster | `priorityclass-high.yaml`, `priorityclass-low.yaml` per BU cluster. Affects scheduling priority for every pod that references them. | -| `manifests/jenkins-filestore-caching/{dev,prd}/{pv,pvc}.yaml` | Per-env | Jenkins build cache PV/PVC backed by GCP Filestore. | -| `manifests/jenkins-gcs-caching/{pv,pvc,sc-gcs}.yaml` | Per-env | Jenkins GCS-backed cache. | -| `manifests/jfrog-filestore-data/{dev,prd}/` | Per-env | JFrog data PV/PVC. | - -### Git hooks - -| Hook | Path | Behavior | -|------|------|----------| -| pre-commit, pre-push | `pre-commit-scripts/runner.sh` | Forks every other `*.sh` in the same dir in parallel; fails the commit if any fails. | -| pre-commit, pre-push | `pre-commit-scripts/trufflehog-hook.sh` | Runs `trufflehog git file://. --since-commit HEAD --branch=$(git rev-parse --abbrev-ref HEAD) --json --results=verified`. On a verified hit: prints the finding and POSTs metadata (no raw secret) to `https://observe.meeshogcp.in/api/webhook`. Exit 1 blocks the commit. **Never bypass.** | -| pre-commit, pre-push | `pre-commit-scripts/cac-validate.sh` | Gated on `configs/` paths in the staged diff. This repo has no `configs/`, so it always early-exits. Documented for completeness. | -| pre-commit, pre-push | `pre-commit-scripts/yaakhook.sh` | Gated on `api-collections/` paths. This repo has none — early-exits. | -| post-commit | `post-commit-scripts/runner.sh` → `commit-metric.sh` | Two-phase: synchronous `start` writes a temp file with commit hash + repo info, spawns detached `continue` background process. `continue` polls the local Cursor SQLite DB up to 120 s for `aiCodeTracking.recentCommit.commitHash` to match HEAD, then POSTs the AI line-edit metrics to `https://cursor-server.meeshogcp.in/api/v1/add-commit-metrics`. Skips rebase/merge/cherry-pick commits. Always exits 0 — never blocks the hook. | - -### Configuration touch points - -There is no application config to touch. The only env-equivalent layer is `helm-overrides///custom-values.yaml` — every cluster × application pair is its own configuration unit. The CAC API at `https://observe.meeshogcp.in/api/cac/repos` would gate config schema validation, but this repo isn't on that allowlist. - -### Critical invariants & gotchas - -- **`helm-templates/` is mostly upstream code.** Most charts here are `helm pull`-ed copies of upstream charts (Bitnami, ArgoProj, VictoriaMetrics, Grafana). Editing a `templates/*.yaml` inside one of these is editing upstream — easy to forget on the next chart bump. Treat `helm-templates//` as read-only unless you are explicitly forking; if you fork, document why in the chart's `README.md`. -- **Each cluster is unique on `nodeSelector` / `tolerations` / `computeClass`.** See `contour-nodeselector-tolerations-summary.md` at the repo root. GKE Autopilot clusters use `cloud.google.com/compute-class` keys; standard clusters use `dedicated:` keys. Copying a values block from one cluster to another without rewriting these schedules pods on the wrong nodes — or pending forever. -- **Versioned siblings are intentional, not duplicates.** `argo-cd` vs `argo-cd-green`, `contour` vs `contour-v1.33.3`, `keda` vs `keda-2.17.1`, `opentelemetry-collector` vs `-latest`, `sonarqube` vs `sonarqube-old`. Don't "consolidate" them. They support blue-green chart upgrades — both versions may be live during a migration. -- **`fullnameOverride` is load-bearing in dataplane overrides.** `db-*` cluster values use `fullnameOverride: -dbc--prd` to keep release names stable across re-installs. Don't change these — Service DNS, PVC binding, and Argo Application names depend on them. -- **Argo CD reconciles from `main`.** A merge to `main` is a deploy. There is no staging branch — review the PR as if it ships to production, because it does. -- **`devops-infra-argo-config` is the routing layer.** A new chart in `helm-templates/` does nothing until an `Application` referencing it is added to the sister repo. A new cluster directory in `helm-overrides/` does nothing until an `ApplicationSet` covers it. Pair the two-repo change. - - diff --git a/docs/global/AGENT_BOUNDARIES.md b/docs/global/AGENT_BOUNDARIES.md deleted file mode 100644 index 7ed5b50..0000000 --- a/docs/global/AGENT_BOUNDARIES.md +++ /dev/null @@ -1,123 +0,0 @@ -# AGENT_BOUNDARIES.md - -> **Scope:** operations agents may perform on the `devops-infra-helm-charts` repo. -> **Companion docs:** [SANCTITY_RULES.md](SANCTITY_RULES.md) (hard rules), [coding-guidelines/helm-values.md](coding-guidelines/helm-values.md) (style). - -This document classifies every operation in this repo into Layer 1 / 2 / 3 with explicit blast radius and approval requirements. **An agent operating in this repo MUST consult this file before any write.** - ---- - -## The 3-Layer Model (recap) - -| Layer | Agent action | Safety gate | -|-------|--------------|-------------| -| **Layer 1 — Agent-Writable** | Generate diff, open PR. Do **not** apply directly to clusters. | PR review + Argo CD UI Sync click. | -| **Layer 2 — Agent-Readable (Advisory)** | Research, analyse, recommend. Human executes. | Human review + manual execution. | -| **Layer 3 — Agent-Blocked** | Refuse the write. Explain why. Cite this doc. | None applicable. | - ---- - -## Per-operation classification - -### Layer 1 — Agent-Writable (this repo's normal operating range) - -All edits land via PR. A merge to `main` is then reconciled by Argo CD per cluster — most infra Applications use manual sync ([ADR-A5](../../wiki/analyses/ADR-A5-manual-sync-default-for-infra.md)), so the workload deploy is a separate human Sync click. - -| Operation | File(s) touched | Blast radius | Required approvers | -|-----------|-----------------|--------------|--------------------| -| Edit `custom-values.yaml` for one app on one cluster | `helm-overrides///custom-values.yaml` | One Helm release on one cluster | Service / app owner | -| Add a new app under an existing cluster | new dir + files in `helm-overrides///` | One new release | App owner + cluster owner; pair with Argo `Application` PR in sister repo | -| Add a sidecar raw manifest alongside an existing app | `helm-overrides///.yaml` | Adds Kubernetes resources rendered alongside the Helm release | App owner | -| Add a new override file under an existing app dir | new file in `helm-overrides///` | Layered into the same release if the Argo Application's `valueFiles` covers it | App owner | -| Append to an existing chart's `Chart.yaml` `dependencies[]` | `helm-templates//Chart.yaml` + refresh `Chart.lock` via `helm dependency update` | Affects every consumer of that chart | Platform team | -| Bump a `dependencies[].version` in `Chart.yaml` | `helm-templates//Chart.yaml` + `Chart.lock` | Same | Platform team — see [update-chart-version](../platform/procedures/update-chart-version.md) | -| Add a new chart sibling for blue-green migration | new dir `helm-templates/-/` | Migration target only; old sibling stays live | Platform team — see [blue-green-chart-migration](../platform/procedures/blue-green-chart-migration.md) | -| Edit the per-cluster Contour scheduling matrix | `contour-nodeselector-tolerations-summary.md` | Documentation; no runtime effect | Platform team | - -### Layer 1 — HIGH RISK (write allowed, but require explicit approval and detailed rationale in PR) - -| Operation | Why it's high risk | -|-----------|--------------------| -| Edit `helm-templates//templates/` or `helm-templates//values.yaml` | Most charts here are vanilla upstream pulled via `helm pull`. Edits silently fork the chart and get clobbered on the next upstream sync. **Only allowed if the fork is intentional and documented in that chart's `README.md` — see [fork-upstream-chart](../platform/procedures/fork-upstream-chart.md).** | -| Delete a versioned sibling chart (`-green`, `-vX.Y.Z`, `-latest`, `-old`) | The variant is a blue-green migration target. Both versions may be live simultaneously. **Confirm zero references in `github.com/Meesho/devops-infra-argo-config` before deleting.** | -| Edit `manifests/storageclass/*.yaml` | Cluster-wide singleton; affects every PVC. **Platform-team review required.** | -| Edit `manifests/priorityclass//*.yaml` | Affects scheduling priority for every pod that references the class. Platform-team review. | -| Edit `manifests/{jenkins-filestore-caching,jenkins-gcs-caching,jfrog-filestore-data}/{dev,prd}/` | Per-env stateful PV/PVCs; wrong reclaim policy can drop CI build caches or JFrog binary data. Platform-team review. | -| Add a new cluster directory under `helm-overrides/` | Creates a new deployment target. **Pair with the cluster's `ApplicationSet` change in the sister repo. Per-cluster `nodeSelector` / `tolerations` / `computeClass` MUST be written from scratch, not copied from another cluster ([SANCTITY_RULES R5](SANCTITY_RULES.md)).** | -| Change `fullnameOverride` in any `custom-values.yaml` | Service DNS, PVC binding, ConfigMap/Secret references, and downstream Argo Application names depend on it being stable. **Almost always wrong to touch.** | -| Bulk find-replace across cluster directories ("normalise" labels, image tags, etc.) | The whole point of per-cluster overrides is divergence. Cross-cluster cleanup is its own PR with its own scope ([SANCTITY_RULES R6](SANCTITY_RULES.md)). | -| Pin an image tag to a registry outside `asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/` | Production overrides go through Meesho's Artifact Registry mirror. Direct Docker Hub / Quay pulls are a supply-chain and rate-limit risk. | - -### Layer 2 — Agent-Readable (advisory only) - -Some operations a user might ask for touch systems an agent can analyse but not change. - -| Operation | Why advisory | -|-----------|--------------| -| "Sync `` now in Argo CD" | The Sync click is intentional human action. **Recommend the command (`argocd app sync `) or the Argo CD UI path; do not execute.** | -| "Restart pods for ``" | kubectl operation against a workload cluster. Recommend `kubectl rollout restart deploy/ -n `; do not execute. | -| "Why is this pod pending?" | Live cluster state. Recommend `kubectl describe pod` and walk [pod-pending-scheduling.md](../platform/runbooks/pod-pending-scheduling.md); do not infer. | -| "Why did the chart not render?" | Reproduce locally with `helm template`; recommend the fix. Do not push without PR. | -| "Add an alert rule for ``" | Alert rules are in `victoria-metrics-alert*` chart values *here*, but routing/notifier config is elsewhere (Pulse / Slack webhook secrets). Recommend the right file; pair with the routing change. | -| "What's broken on the GKE cluster itself?" | Cluster-level GCP / GKE issues are out of scope for this repo. Recommend the Terraform repo or the platform team. | -| "Reconcile the Terraform drift for the cluster" | Out of scope — that's `terraform-gcp-infra`. Refuse + redirect. | - -### Layer 3 — Agent-Blocked (refuse + explain) - -| Operation | Why blocked | What would unblock | -|-----------|-------------|--------------------| -| Hand-edit `repository.yaml` | Owned by `registry-bootstrap` automation; manual edits are overwritten on the next bootstrap run. | Update the upstream registry that feeds registry-bootstrap. | -| Bypass pre-commit hooks (`--no-verify`, `git commit -n`, removing the hook) | TruffleHog is the last-line secret scan. ([SANCTITY_RULES R4](SANCTITY_RULES.md)) | If a hit is a known false positive, confirm with the platform/security team in writing first. | -| Push directly to `main` | Branch protection. ([SANCTITY_RULES R1](SANCTITY_RULES.md)) | (Never legitimate.) | -| Force-push to `main` | Same. | (Never legitimate.) | -| Run `helm install`, `helm upgrade`, or `kubectl apply` against any workload cluster | This is GitOps; in-cluster mutation creates drift Argo CD will reconcile away. | Use the Argo CD UI Sync flow, or the cluster's incident-response toolset. | -| Curl / probe / interact with `int.meesho.int`, `prd.meesho.int`, `int.mrouter.int`, `prd.mrouter.int`, or any workload-traffic `*.meeshogcp.in` host | Production traffic surfaces. ([SANCTITY_RULES R3](SANCTITY_RULES.md)) | (Pre-commit hook telemetry to `observe.meeshogcp.in` is automated; that is hook infrastructure, not agent action.) | -| Edit Argo `Application` / `ApplicationSet` manifests | They live in `github.com/Meesho/devops-infra-argo-config`. | Open a PR there. | -| Modify a chart whose `helm-templates//` is vanilla upstream, without a documented fork rationale | An accidental fork is silently clobbered on the next sync. | Follow [fork-upstream-chart](../platform/procedures/fork-upstream-chart.md) and document the fork in the chart's `README.md`. | -| Copy a `custom-values.yaml` from one cluster to another verbatim | Per-cluster `nodeSelector` / `tolerations` / `computeClass` differ. ([SANCTITY_RULES R5](SANCTITY_RULES.md)) | Author the override from scratch using the per-cluster matrix in [contour-nodeselector-tolerations-summary.md](../../contour-nodeselector-tolerations-summary.md) and sibling files. | - ---- - -## Cross-cuts: things to verify on **every** Layer 1 PR - -Pre-commit hooks here are minimal — TruffleHog covers secrets; CAC and Yaak are no-ops because their gating paths don't exist in this repo. The agent is the next line of defence. On every PR, mentally run this checklist: - -1. **Surgical scope.** The PR touches only the cluster × app the task asked for. No drive-by edits to neighbours. -2. **Per-cluster scheduling rewritten, not copied.** If the change involves `nodeSelector` / `tolerations` / `computeClass`, every cluster has its own topology. Copy-and-rename is the most common silent bug. ([contour-nodeselector-tolerations-summary.md](../../contour-nodeselector-tolerations-summary.md)) -3. **`fullnameOverride` unchanged.** Unless the explicit headline of the PR is a release-name migration. -4. **Image tags pin to Meesho's GAR mirror** (`asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/`), not Docker Hub. -5. **Versioned siblings preserved.** A delete or rename of a `-green` / `-latest` / `-vX.Y.Z` directory only proceeds after grepping `github.com/Meesho/devops-infra-argo-config` for references. -6. **Chart.yaml + Chart.lock move together.** A `dependencies[].version` bump without a refreshed lockfile is incomplete. -7. **Sister repo paired (where required).** A new app under a cluster needs a matching `Application` in the sister repo. A new cluster directory needs a matching `ApplicationSet` (or per-cluster Application set) in the sister repo. Note the pairing in the PR description. -8. **No `*` or `latest` image tags introduced.** Argo CD's manual-sync default does not rescue you from a pulled-out-from-under-you image. -9. **`manifests/` left alone unless the PR's headline says so.** `storageclass/` and `priorityclass//` are cluster-wide singletons. -10. **TruffleHog passed.** Never bypass. - ---- - -## Approval requirements summary - -| Change category | Reviewer required | CMR required? | -|-----------------|-------------------|---------------| -| Single-cluster `custom-values.yaml` edit (no scheduling, no `fullnameOverride`) | App owner | Per BU policy | -| Add a new app override under an existing cluster | App owner + cluster owner | **Yes** | -| Add a new cluster directory | Platform team | **Yes** | -| `Chart.yaml` `dependencies[].version` bump | Platform team | **Yes** for prod-fleet charts (Argo CD, Contour, VictoriaMetrics, ingress) | -| Intentional fork of `helm-templates//templates/` | Platform team — multi-reviewer | **Yes** | -| Delete a versioned chart sibling | Platform team — confirm zero sister-repo references | **Yes** | -| Edit `manifests/storageclass/` or `priorityclass//` | Platform team | **Yes** | -| Pre/post-commit hook script change | Platform team + security (if hook scope changes) | Per CMR matrix | -| `fullnameOverride` change | Platform team — multi-reviewer | **Yes** — emergency-only | - -CMR = Change Management Request. Per Meesho process; not enforced in-repo, applied at org level. - ---- - -## Escalation - -If a request falls outside this matrix or you can't classify it cleanly: - -1. Refuse the write. -2. Cite this document + the row that matches (or explain why no row matches). -3. Suggest the human asks the platform team or files a CMR. -4. Do not improvise around the boundary. diff --git a/docs/global/SANCTITY_RULES.md b/docs/global/SANCTITY_RULES.md deleted file mode 100644 index 50c84da..0000000 --- a/docs/global/SANCTITY_RULES.md +++ /dev/null @@ -1,122 +0,0 @@ -# SANCTITY_RULES.md - -> Non-negotiable rules for the `devops-infra-helm-charts` repo. -> Read alongside [AGENT_BOUNDARIES.md](AGENT_BOUNDARIES.md) (per-operation Layer map) and [coding-guidelines/helm-values.md](coding-guidelines/helm-values.md) (style). - -These are the rules whose violation is a process incident, not a clever shortcut. Each one is the result of a real failure mode (or proximity to one). - ---- - -## R1 — `main` is production - -Argo CD on every cluster reconciles from `main`. There is no staging branch. A merge is a deploy event. - -- **No experimentation on `main`.** Always work on a feature/fix branch and open a PR. -- **No force-push to `main`** (enforced at GitHub org level). -- **No "I'll just amend that" after merge.** A new PR is the only path forward. - -## R2 — Sister repo is the routing layer; this repo is the values layer - -`devops-infra-helm-charts` holds *what gets installed and with what values*. `github.com/Meesho/devops-infra-argo-config` holds *where it gets routed* (the Argo `Application` / `ApplicationSet` manifests). - -- **A new chart in `helm-templates/` does nothing** until an `Application` referencing it lands in the sister repo. -- **A new cluster directory in `helm-overrides/` does nothing** until an `ApplicationSet` (or per-cluster Application set) covers it. -- **Pair the two-repo change.** Cite the sister-repo PR in the description here, and vice versa. - -## R3 — Production traffic surfaces are off-limits - -Agent-initiated requests to `int.meesho.int`, `prd.meesho.int`, `int.mrouter.int`, `prd.mrouter.int`, and workload `*.meeshogcp.in` services are forbidden. Any accidental call can affect live traffic or data. - -- **Never `curl`, `WebFetch`, query, or otherwise probe** these endpoints. -- The pre-commit hook telemetry endpoints (`observe.meeshogcp.in/api/webhook`, `cursor-server.meeshogcp.in/api/v1/...`) are *automated hook infrastructure*, not agent-initiated traffic. They are not a precedent for agent calls. - -## R4 — Pre-commit hooks must pass - -`pre-commit-scripts/runner.sh` invokes TruffleHog (verified-secret scan with webhook telemetry). CAC and Yaak hooks exist but are no-ops here (their gating paths — `configs/` and `api-collections/` — don't exist in this repo). TruffleHog is the last-line secret scan. - -- **Never bypass** with `--no-verify`, `git commit -n`, or by removing the hook. -- **Never weaken a hook to "just get this through"** — fix the upstream cause. -- A verified TruffleHog hit means a secret is staged. Real secrets belong in `external-secrets` (per-cluster), backed by GCP Secret Manager / Vault — not in `custom-values.yaml`. - -## R5 — Per-cluster scheduling is bespoke; never copy across clusters - -Each cluster has its own node-pool topology. Some clusters use GKE Autopilot's `cloud.google.com/compute-class:` keys (`k8s-central-prd-ase1`, `k8s-dsgpu-prd-ase1`, `k8s-shared-int-ase1`); standard clusters use `dedicated:` keys. Per-Contour-instance, per-cluster mappings are recorded in [contour-nodeselector-tolerations-summary.md](../../contour-nodeselector-tolerations-summary.md). - -- **Never copy `nodeSelector` / `tolerations` / `computeClass` blocks between clusters** without rewriting them from scratch against the destination cluster's topology. -- **Always read the matrix** before editing any Contour values; cross-reference sibling cluster files for non-Contour scheduling. -- Wrong values strand pods on wrong nodes or leave them `Pending` indefinitely. - -## R6 — Surgical edits only - -The repo has ~30 cluster directories × dozens of apps each. The whole point of per-cluster overrides is divergence — accumulated, deliberate, often for tiered traffic or specialised hardware. - -- **Never "normalise" values across clusters in the same PR as a feature change.** Cross-cluster clean-ups are their own PRs with their own scope and CMRs. -- **Never auto-update labels, image tags, or comments** on apps you weren't asked to touch. -- **One change-type per PR.** Reviewer cognition matters; rollback granularity matters. - -## R7 — `helm-templates//` is mostly upstream - -Most charts here are vanilla upstream pulled via `helm pull`. The local `Chart.yaml` is often a thin wrapper that declares the upstream chart as a dependency. Editing a `templates/*.yaml` or the chart's own `values.yaml` silently forks the chart, and the fork is clobbered on the next upstream sync. - -- **Never edit `helm-templates//templates/` or `values.yaml` casually.** If a fork is intentional, follow [fork-upstream-chart](../platform/procedures/fork-upstream-chart.md) and document the reason in the chart's `README.md`. -- **Never bump `Chart.yaml` `dependencies[].version`** without (a) reading the upstream changelog, (b) running `helm dependency update` to refresh `Chart.lock`, and (c) calling out the bump in the PR description. - -## R8 — Versioned chart siblings stay live during migrations - -`` ↔ `-green` (blue-green), `` ↔ `-vX.Y.Z` (pinned upgrade target), `` ↔ `-latest` (work-in-progress), `` ↔ `-old` (retired but still referenced). Both directories may be referenced by Argo Applications during the migration window. - -- **Never delete a versioned sibling** without confirming zero references in `github.com/Meesho/devops-infra-argo-config`. -- **Never "consolidate" siblings** into one chart in a maintenance PR. The split is intentional ([ADR-A2](../../wiki/analyses/ADR-A2-blue-green-sibling-pattern.md)). - -## R9 — `fullnameOverride` is load-bearing - -Helm's `fullnameOverride` controls the name of every released Service, Deployment, StatefulSet, ConfigMap, Secret, and PVC. Service DNS, PVC binding, ConfigMap references in other apps, and Argo Application names downstream all depend on it being stable. - -- **Never change `fullnameOverride`** in any `custom-values.yaml`. -- A release-name migration is its own headline-of-the-PR change with platform-team approval, a documented before/after map, and an explicit reason. - -## R10 — `manifests/` is cluster-wide singletons - -`manifests/storageclass/*.yaml` is repo-global; a wrong StorageClass affects every PVC on every cluster that consumes it. `manifests/priorityclass//*.yaml` is per-cluster but cluster-wide; a wrong PriorityClass changes scheduling priority for every pod that references it. - -- **Never edit `manifests/storageclass/`** or `manifests/priorityclass//` without platform-team review. -- **Never delete a StorageClass referenced by an existing PVC** — Kubernetes will not remove the StorageClass while bindings remain, but new PVCs will fail to provision. - -## R11 — Image tags go through Meesho's Artifact Registry mirror - -Production overrides pin images to `asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/`, not upstream Docker Hub or Quay. The mirror exists for supply-chain control and rate-limit isolation. - -- **Never introduce a Docker Hub / Quay / GCR / ECR upstream tag** in a production override. -- **Never use `:latest` or unpinned tags** in production overrides; an Argo CD reconcile is not the same as a controlled rollout. - -## R12 — `repository.yaml` is automation-owned - -Generated by `registry-bootstrap`. Hand edits will be silently overwritten on the next bootstrap run. - -- **Never hand-edit `repository.yaml`.** If owner data is wrong, fix it in the upstream registry that feeds registry-bootstrap. - -## R13 — Branch protection trumps everything - -Direct push to `main` is blocked at the GitHub org level. PRs go through code review. - -- **Never propose workflows that bypass branch protection.** If a hotfix is genuinely urgent, the path is "open a PR with a `hotfix/*` branch and request emergency review," not "force-push to main." - -## R14 — Pre-existing schema/label drift is not yours to fix - -Some clusters have inconsistencies in label values, indentation, key ordering, or comment style — accumulated technical-debt items. Some apps lack values keys their newer siblings have. Some clusters use long BU names (`supply`) while others use short (`supl`). - -- **Never normalise label or BU values in unrelated PRs.** Cross-cutting clean-ups are their own PRs with their own scope and CMRs. -- **Never reformat a YAML file in passing.** Diffs full of indentation churn drown out the real change. - ---- - -## What "non-negotiable" means - -Each rule above has a documented reason and is the result of a real failure mode (or proximity to one). If you find yourself wanting to break a rule, the path is: - -1. Write up the case in a doc (or PR description) explaining what would change and why. -2. Loop in the platform team for review. -3. If the rule should change, change *the rule first* in this file (with a PR to update it), then act. -4. If the rule should not change, find another way. - -A merge that violates a rule here is a process incident, not a clever shortcut. diff --git a/docs/global/agent-operations-guide.md b/docs/global/agent-operations-guide.md deleted file mode 100644 index 7413edf..0000000 --- a/docs/global/agent-operations-guide.md +++ /dev/null @@ -1,82 +0,0 @@ -> Per AI Blitz Plan §global. Layer: 1. Repo: devops-infra-helm-charts. - -# Agent Operations Guide - -A meta-guide for any AI agent (or human contributor wearing the agent hat) doing work in this repo. Read this first; it tells you which other docs to load and in what order. - -## Pre-flight (always, every task) - -Before reading any task-specific file: - -1. **Read the repo-root `CLAUDE.md`.** It is the authoritative source for the NEVER-DO list, cluster naming, versioned siblings, multi-Contour pattern, and the Layer constraint summary. Everything else in `docs/` derives from it. -2. **Read [AGENT_BOUNDARIES.md](./AGENT_BOUNDARIES.md)** to confirm which Layer the operation lives in. -3. **Read [SANCTITY_RULES.md](./SANCTITY_RULES.md)** to confirm the operation is not on the don't-touch list. -4. **Read [`wiki/entities/DevOps Infra Helm Charts.md`](../../wiki/entities/DevOps%20Infra%20Helm%20Charts.md)** for the conceptual model (this repo as values store; sister repo as routing layer). -5. **Skim [`docs/architecture.md`](../architecture.md)** if the task touches more than one chart or cluster. - -If the task targets a specific chart family (Contour, Vault, Argo CD, observability stack), also load the relevant coding guideline before editing: -- [coding-guidelines/helm-values.md](./coding-guidelines/helm-values.md) — values-file conventions -- [coding-guidelines/argocd.md](./coding-guidelines/argocd.md) — Argo CD interaction model -- [coding-guidelines/observability.md](./coding-guidelines/observability.md) — VM / Mimir / Loki / Tempo / Grafana - -## Authoring loop - -The supported edit path for almost every task in this repo: - -1. **Branch off `main`.** Never push to `main`. The branch name should describe the cluster × app being touched. -2. **Edit the single targeted file** under `helm-overrides///custom-values.yaml` (or sibling `.yaml`). No drive-by edits to other apps in the same directory. No cross-cluster "normalization" in the same PR. -3. **Dry-run with `helm template`** to confirm the values render. Use the command from the root `CLAUDE.md` Quick reference table: - ``` - helm template helm-templates/ -f helm-overrides///custom-values.yaml - ``` -4. **Commit.** The pre-commit hooks run automatically: - - **TruffleHog** — secret scan, blocking. NEVER bypass. See [claude/08-pre-commit-and-hooks.md](../../claude/08-pre-commit-and-hooks.md). - - CAC and Yaak hooks are gated/no-op on this repo. -5. **Open a PR.** Pair with a sister-repo PR if a new Argo Application is being introduced. -6. **Reviewer + Argo CD UI Sync are the safety gates.** Once the PR is merged to `main`, Argo CD on the target cluster reconciles. Per-cluster `syncPolicy` (auto vs manual) is set in the sister repo `Meesho/devops-infra-argo-config`, not here. - -See [platform/procedures/onboard-app-to-cluster.md](../platform/procedures/onboard-app-to-cluster.md) and [platform/procedures/update-chart-version.md](../platform/procedures/update-chart-version.md) for two of the most common authoring loops. - -## When to refuse - -Refuse the operation outright if it falls into one of these categories. Cite [SANCTITY_RULES.md](./SANCTITY_RULES.md) in the refusal: - -- Direct push to `main`, or any `--force` push. -- Bypassing the pre-commit hook (`--no-verify`, `git commit -n`). -- Curl / probe / query against any production endpoint (`*.meesho.int`, `*.mrouter.int`, `*.meeshogcp.in`). -- Edit to `repository.yaml` (owned by `registry-bootstrap`). -- Edit to an Argo `Application` / `ApplicationSet` manifest (lives in sister repo). -- Cross-cluster copy-paste of `nodeSelector` / `tolerations` / `computeClass` without rewrite. -- Deletion of a versioned-sibling chart without confirming zero sister-repo references. - -## When to escalate - -If the operation is potentially valid but exceeds Layer-1 authority — chart fork, dep bump with breaking changes, StorageClass/PriorityClass edit, Vault HA work — stop and escalate per [escalation-matrix.md](./escalation-matrix.md). The escalation matrix maps each situation to an owner and a channel. - -## Tooling commands (cheat sheet) - -The repo-root `CLAUDE.md` Quick reference table is the source of truth. Reproduced here for convenience: - -| Task | Command | -|------|---------| -| Install pre-commit hooks | `pre-commit install --hook-type pre-commit --hook-type pre-push --hook-type post-commit` | -| Re-run pre-commit on staged changes | `pre-commit run` | -| Render a chart locally | `helm template helm-templates/ -f helm-overrides///custom-values.yaml` | -| Refresh subchart deps | `helm dependency update helm-templates/` | -| Diff against live release | `helm diff upgrade helm-templates/ -f helm-overrides///custom-values.yaml` | -| Lint a chart | `helm lint helm-templates/` | -| Find which clusters override an app | `find helm-overrides -maxdepth 2 -type d -name ''` | - -## Output discipline - -- Generate diffs and PRs; do not apply directly to clusters. In-cluster mutation is incident response, not authoring. -- Surgical edits only. Touch the cluster × application asked for; leave the rest. -- Cross-link reasoning to ADRs in [`wiki/analyses/`](../../wiki/analyses/) where helpful. - -## See also - -- [AGENT_BOUNDARIES.md](./AGENT_BOUNDARIES.md) -- [SANCTITY_RULES.md](./SANCTITY_RULES.md) -- [escalation-matrix.md](./escalation-matrix.md) -- [coding-guidelines/helm-values.md](./coding-guidelines/helm-values.md) -- [../architecture.md](../architecture.md) diff --git a/docs/global/coding-guidelines/argocd.md b/docs/global/coding-guidelines/argocd.md deleted file mode 100644 index 749bfea..0000000 --- a/docs/global/coding-guidelines/argocd.md +++ /dev/null @@ -1,67 +0,0 @@ -> Per AI Blitz Plan §global. Layer: 1. Repo: devops-infra-helm-charts. - -# Coding Guideline — Argo CD interaction model - -This repo holds **values** and **cached charts**. It does NOT own Argo CD `Application` or `ApplicationSet` manifests. Those live in the sister repo: - -> [`github.com/Meesho/devops-infra-argo-config`](https://github.com/Meesho/devops-infra-argo-config) - -A merge to `main` here is a deploy event for every cluster whose Argo Application points at a path in this repo. The Argo `Application` defines the routing (which path on which cluster, sync policy, retry, prune); we define what that path renders to. - -## Hard separation - -| Concern | Owns it | -|---------|---------| -| `helm-templates//` (cached/forked chart) | this repo | -| `helm-overrides///custom-values.yaml` (per-cluster values) | this repo | -| `manifests/storageclass/`, `manifests/priorityclass//` (singletons) | this repo | -| Argo `Application` (cluster, path, repoURL, targetRevision, destination namespace) | **sister repo** | -| Argo `ApplicationSet` (cluster generators, templating fan-out) | **sister repo** | -| `syncPolicy.automated.{prune,selfHeal}` decision | **sister repo** | -| `syncPolicy.syncOptions` (CreateNamespace, ServerSideApply) | **sister repo** | -| Sync waves / hooks via `argocd.argoproj.io/sync-wave` annotations | this repo (when expressed inside chart templates or raw sidecar manifests) | - -## What this means for the agent - -- **Never** add or edit a file matching `Application*.yaml` / `ApplicationSet*.yaml` here. If the task asks for one, redirect to the sister repo. See [escalation-matrix.md](../escalation-matrix.md) row 5. -- When introducing a **new app** to a cluster, the change is a **paired PR**: (a) a PR here adding `helm-overrides///custom-values.yaml`, and (b) a PR in the sister repo adding the matching `Application` manifest. Both must merge before the app deploys. -- When introducing a **new cluster**, the paired PR in the sister repo updates the `ApplicationSet` cluster generator. See [../platform/procedures/onboard-new-cluster.md](../../platform/procedures/onboard-new-cluster.md). -- When **removing** an app, deboard the Argo Application first (sister repo), let Argo prune, then remove the override directory here. See [../platform/procedures/deboard-app.md](../../platform/procedures/deboard-app.md). - -## Sync policy: where the decision lives - -`syncPolicy.automated.prune` and `syncPolicy.automated.selfHeal` live in the sister repo's `Application` spec. Convention on the platform: - -- **Manual sync default** for infra components on prod clusters. Sync is a deliberate human click after a merge. Rationale documented in [`wiki/analyses/ADR-A5-manual-sync-default-for-infra.md`](../../../wiki/analyses/ADR-A5-manual-sync-default-for-infra.md). -- Auto-sync is reserved for low-risk leaf components (e.g., `kube-state-metrics`, monitoring agents) where reconciliation drift is benign. - -Agents editing values here should assume **the sync click is the safety gate**. A merged PR is not yet deployed. - -## Validating values before merge - -The agent's responsibility is that the values **render** correctly. Argo CD will materialize the rendered output via `helm template`-equivalent server-side. Use the same command locally: - -``` -helm template helm-templates/ \ - -f helm-overrides///custom-values.yaml -``` - -If the chart has subchart dependencies (`Chart.yaml` `dependencies:`), run `helm dependency update helm-templates/` before templating, otherwise render will fail with `found in Chart.yaml, but missing in charts/ directory`. - -For raw-manifest sidecars (`.yaml` files in the override dir), validate with `kubectl apply --dry-run=client -f `. See [../platform/schemas/raw-manifest-sidecar-schema.md](../../platform/schemas/raw-manifest-sidecar-schema.md). - -## Common failure modes - -| Symptom | First read | -|---------|-----------| -| `Sync` button click results in `OutOfSync` that won't resolve | [../platform/runbooks/argocd-sync-failure.md](../../platform/runbooks/argocd-sync-failure.md) | -| Pods land but stay `Pending` | [../platform/runbooks/pod-pending-scheduling.md](../../platform/runbooks/pod-pending-scheduling.md) | -| Ingress 5xx after a Contour values change | [../platform/runbooks/ingress-down.md](../../platform/runbooks/ingress-down.md) | - -## Cross-references - -- Sister repo: [`Meesho/devops-infra-argo-config`](https://github.com/Meesho/devops-infra-argo-config) -- [helm-values.md](./helm-values.md) -- [observability.md](./observability.md) -- [../escalation-matrix.md](../escalation-matrix.md) -- [`wiki/analyses/ADR-A5-manual-sync-default-for-infra.md`](../../../wiki/analyses/ADR-A5-manual-sync-default-for-infra.md) diff --git a/docs/global/coding-guidelines/helm-values.md b/docs/global/coding-guidelines/helm-values.md deleted file mode 100644 index d32a829..0000000 --- a/docs/global/coding-guidelines/helm-values.md +++ /dev/null @@ -1,167 +0,0 @@ -# Coding Guidelines — Helm values authoring - -> The hard stops live in [../SANCTITY_RULES.md](../SANCTITY_RULES.md); the layer/scope rules in [../AGENT_BOUNDARIES.md](../AGENT_BOUNDARIES.md). This file is "how to write the values YAML well" — style, conventions, and the recurring footguns the linter doesn't catch. - ---- - -## File-level conventions - -### Where files go - -| Path | Meaning | -|------|---------| -| `helm-overrides///custom-values.yaml` | The primary Helm values file Argo's `valueFiles` references. **Default name. Do not invent alternatives.** | -| `helm-overrides///.yaml` | Sidecar raw manifests applied alongside the Helm release. Common shapes: `computeclass/-cc.yaml`, `external-dns-services/.yaml`, `elastic-cluster/argo-launch.yaml`, `mimir-distributed/alertmanager_config.yaml`. | -| `helm-templates//Chart.yaml` | Chart manifest; usually a thin wrapper declaring an upstream dep. | -| `helm-templates//values.yaml` | The chart's own defaults — rarely edited; treat as upstream. | -| `helm-templates//templates/` | Manifest templates — vanilla upstream unless intentionally forked. **Don't edit casually.** | - -### Cluster directory naming - -The cluster directory's name is a contract — it must match the Kubernetes cluster name registered in Argo CD. Conventions: - -| Pattern | Use | -|---------|-----| -| `k8s--prd-ase1` | Standard GKE prod cluster, BU-owned (AWS-style naming retained). | -| `k8s--prd-ase1c` | GCP zone-c twin. | -| `k8s-shared-int-ase1` | Shared int (pre-prod) cluster. The only non-prod cluster. | -| `k8s-aurva-prd-ase1` | Aurva integration. | -| `db--...` | Auto-named dataplane / data-tier clusters. Minimal override sets (typically `kube-state-metrics` + `victoria-metrics-agent`). | -| `k8s-supply-dev-ase1` | The lone dev/sandbox cluster. | - -### App directory naming - -Inside `helm-overrides//`, each subdirectory is one Argo Application = one Helm release. - -- One Helm release per directory. -- Multiple Contour instances per cluster is the norm — `contour-external`, `contour-internal-0`, `contour-internal-1`, `contour-internal-intra-{0,1}`. Each maps to a different node pool / dedicated taint or compute class. **They are separate releases — don't merge them.** -- Versioned siblings (`argo-cd` ↔ `argo-cd-green`, `keda` ↔ `keda-2.17.1`) live as parallel directories under `helm-templates/`, but their per-cluster overrides typically live under one directory until the migration cuts over. See [blue-green-chart-migration](../../platform/procedures/blue-green-chart-migration.md). - ---- - -## Field-level conventions (`custom-values.yaml`) - -### Image references - -- **Always pin images to Meesho's Artifact Registry mirror** for production: - ```yaml - image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/ - tag: - ``` -- **Never use `:latest` or unpinned tags** in production overrides. -- **Never reference Docker Hub, Quay, GCR upstream, or ECR directly** in a production override. Mirror it via the platform team's image-pull workflow first. - -### `fullnameOverride` - -- **Set it once, never change it.** Service DNS, PVC binding, ConfigMap references, and downstream Argo Application names depend on stability. ([SANCTITY_RULES R9](../SANCTITY_RULES.md)) -- For dataplane (`db-*`) clusters, the convention is `fullnameOverride: -dbc--prd` (e.g. `kube-state-metrics-dbc-dsci-prd`). -- For BU clusters, omit `fullnameOverride` unless the chart's default name collides with another release in the same namespace. - -### `nameOverride` - -Almost never needed. Helm's `-` naming is usually fine. - -### Replica counts and HPA bounds - -- Read the chart's defaults before specifying replicas. Some charts have HPA-managed replicas that conflict with `replicaCount`. -- For HPA-managed releases, set `minReplicas` and `maxReplicas` *and* leave `replicaCount` unset (or set to `null`). -- Don't lower `minReplicas` to zero unless the workload genuinely scales-from-zero. - -### Resource requests and limits - -- **Always set `resources.requests`** for production releases. Without requests, the scheduler treats the pod as best-effort. -- **Set `resources.limits`** unless the chart documentation explicitly recommends omitting them (some sidecars deliberately go limit-less). -- **Don't copy resources from another cluster.** Workload sizing is per-traffic-tier; the supply prd cluster's Contour requests are not the demand prd cluster's. - -### `nodeSelector`, `tolerations`, `affinity`, `topologySpreadConstraints` - -The single biggest source of silent mis-deploys. Per [SANCTITY_RULES R5](../SANCTITY_RULES.md): - -| Cluster type | Key style | Example | -|--------------|-----------|---------| -| GKE Autopilot (`k8s-central-prd-ase1`, `k8s-dsgpu-prd-ase1`, `k8s-shared-int-ase1`) | `cloud.google.com/compute-class` | `nodeSelector: {cloud.google.com/compute-class: contour-internal-0-cc}` | -| Standard GKE | `dedicated:` | `nodeSelector: {dedicated: contour-internal-0}` | -| Most others | `dedicated:` | same | - -Cross-reference [`contour-nodeselector-tolerations-summary.md`](../../../contour-nodeselector-tolerations-summary.md) for the per-cluster Contour matrix. For non-Contour apps, copy from a sibling app on the *same* cluster, not from the same app on a *different* cluster. - -### Probes - -- Always set `livenessProbe` and `readinessProbe` for any long-running container. -- For workloads that take >30 s to warm (Jenkins, JFrog, ClickHouse), bump `initialDelaySeconds` accordingly — not the timeout, not the period. -- A `startupProbe` is the right tool for slow-warm containers; don't fight it with a 600s `initialDelaySeconds` on `livenessProbe`. - -### Persistence - -- StorageClass references go through the cluster-wide singletons in `manifests/storageclass/`: `pd-standard-retain-dr`, `sc-filestore-standard`, `sc-pd-ssd`, `sc-pd-standard`. **Never reference a StorageClass that doesn't exist in `manifests/storageclass/`.** -- For stateful releases, set `persistence.size` explicitly. Default sizes are rarely right. -- Never enable `persistence.enabled: true` without confirming the StorageClass and its retention/reclaim policy. - -### Secrets - -- **Never inline secret values** in `custom-values.yaml`. ([SANCTITY_RULES R4](../SANCTITY_RULES.md)) -- Reference secrets by name: `existingSecret: `, where the secret is materialised by the per-cluster `external-secrets` app from GCP Secret Manager / Vault. -- Most clusters have an `external-secrets/` override directory; if your release needs a secret, the corresponding `ExternalSecret` lives there. - -### Annotations and labels - -- **Add labels conservatively.** Most charts already emit sensible label sets (`app.kubernetes.io/name`, etc.). -- For ingress (`Ingress`, `HTTPProxy`, Contour `Service`), `external-dns` annotations and AWS/GCP load-balancer annotations are normal — copy from a sibling on the same cluster. -- **Don't invent label keys.** If you find yourself adding `meesho.com/`, double-check whether the project already has a convention for it. - ---- - -## Field-level conventions (raw sidecar manifests) - -For `helm-overrides///.yaml` files (no Helm templating, applied as-is): - -- One Kubernetes resource per file unless they are tightly coupled. -- Use `apiVersion: v1` etc. — pin the API version explicitly. -- Set `metadata.namespace` (don't rely on the Argo Application's `destination.namespace` for these). -- For `ComputeClass` / `NodeClass` / `BackendConfig` / GKE-specific resources, sample a sibling cluster's existing file before authoring. -- For `external-dns-services/*.yaml`, the `Service` resource carries `external-dns.alpha.kubernetes.io/hostname` annotations — match the cluster's existing DNS pattern. - ---- - -## YAML style - -- **2-space indent. No tabs.** -- **Use single quotes for `'*'`** and other glob-like strings; bare strings elsewhere where unambiguous. -- **Trailing newline at EOF.** -- **No `---` document separators** unless you genuinely need multi-document YAML (rare in this repo). -- **Don't comment out fields; remove them.** The repo doesn't use commented-out scaffolding. -- **Preserve key order from siblings.** A reordered file is a noisy diff that drowns the real change. -- **Don't reformat unrelated YAML in passing.** ([SANCTITY_RULES R14](../SANCTITY_RULES.md)) - ---- - -## Diff hygiene - -When opening a PR: - -- **One change-type per PR.** Adding a service should not also "normalise labels on three other apps." -- **Keep diffs minimal.** Don't reformat surrounding YAML. -- **Cite the procedure followed** (link to one of `docs/platform/procedures/*.md`) in the PR description. -- **Show the validation you ran** — `helm template`, `yamllint`, sibling-file diff, the kubectl context you ran a `helm diff` against. -- **Pair the sister-repo PR** (`devops-infra-argo-config`) when adding a new app or cluster — link both. - ---- - -## Common mistakes the hooks do **not** catch - -These are the recurring footguns that pre-commit hooks won't flag: - -1. **`nodeSelector` / `tolerations` / `computeClass` copied from the wrong cluster.** Pods stay `Pending`, or schedule on the wrong node pool. -2. **`fullnameOverride` modified.** Downstream Service DNS resolves to nothing. -3. **`spec.source.path` in the sister repo's `Application` not updated** to point at the new chart sibling after a blue-green migration. -4. **Image tag pinned to Docker Hub** or Quay instead of the GAR mirror. -5. **`replicaCount` set on an HPA-managed release.** HPA fights the static count. -6. **`persistence.storageClass` referencing a class that doesn't exist** on this cluster — PVC stays `Pending` forever. -7. **`existingSecret` referencing a secret the per-cluster `external-secrets` app doesn't create.** Pods crashloop on missing env. -8. **`Chart.yaml` `dependencies[].version` bumped without `helm dependency update`.** Argo CD will use the lockfile and silently render the old version. -9. **Edits inside `helm-templates//templates/`** — silently fork the chart; clobbered on next upstream sync. -10. **Sidecar raw-manifest namespace mismatch** with the Helm release's namespace — orphaned resources. - -The agent's job is to be the second pair of eyes on every one of these. diff --git a/docs/global/coding-guidelines/observability.md b/docs/global/coding-guidelines/observability.md deleted file mode 100644 index 6ca51d2..0000000 --- a/docs/global/coding-guidelines/observability.md +++ /dev/null @@ -1,78 +0,0 @@ -> Per AI Blitz Plan §global. Layer: 1. Repo: devops-infra-helm-charts. - -# Coding Guideline — Observability stack overrides - -Conventions for editing values for the observability charts vendored in this repo: - -- `victoria-metrics-cluster` and `victoria-metrics-cluster-latest` -- `victoria-metrics-agent` and `victoria-metrics-agent-latest` -- `vmalert`, `victoria-metrics-operator` -- `mimir-distributed` -- `loki`, `loki-distributed` -- `tempo`, `tempo-distributed` -- `grafana` -- `opentelemetry-collector` and `opentelemetry-collector-latest` -- `kube-state-metrics`, `node-exporter`, `metrics-server` -- `kube-prometheus-stack` (Prometheus alert rules) - -The observability stack is the most rule-heavy domain in the repo: alert rules drive paging, retention drives storage cost, and label cardinality drives both. Treat values changes as *data-pipeline* changes, not static config. - -## Versioned siblings - -`*-latest` siblings exist alongside the stable chart for each VictoriaMetrics and OTel collector chart. Both can be live simultaneously during a blue-green migration. When editing, confirm which sibling the target Argo Application points at (sister repo). See [`wiki/analyses/ADR-A2-blue-green-sibling-pattern.md`](../../../wiki/analyses/ADR-A2-blue-green-sibling-pattern.md) and [../platform/procedures/blue-green-chart-migration.md](../../platform/procedures/blue-green-chart-migration.md). - -## Cardinality discipline - -Series cardinality on VictoriaMetrics / Mimir is the dominant cost driver. Before adding any of the following, ensure the new label is bounded: - -- New `extraLabels` / `externalLabels` on `victoria-metrics-agent`. -- New `relabel_configs` that emit a label sourced from a high-cardinality metric source (pod name, request path, user id, request id). -- New scrape targets in `additionalScrapeConfigs`. - -If a label can take more than ~few-hundred distinct values, drop it or aggregate before storage. - -## PromQL / alert-rule validation - -Alert rules live in `kube-prometheus-stack`, `vmalert`, and Mimir ruler config. Before merging: - -1. Render with `helm template`: - ``` - helm template prom helm-templates/kube-prometheus-stack \ - -f helm-overrides//kube-prometheus-stack/custom-values.yaml - ``` -2. Extract the `PrometheusRule` objects and validate PromQL with `promtool check rules ` (Prometheus tooling) or `vmalert -dryRun -rule=` for vmalert-specific rules. -3. Confirm the rule's `for:` window and `severity` label match the cluster's PagerDuty routing — getting this wrong silently swaps which on-call gets paged. - -## Retention and tenancy - -- VictoriaMetrics `vmstorage.retentionPeriod` is per-cluster. Increasing it grows the PVC; never increase without confirming PVC headroom and matching PVC `resources.requests.storage`. -- Loki and Mimir multi-tenancy is keyed on the `X-Scope-OrgID` header. Tenant lists live in cluster-specific overrides; do not assume tenants match across clusters. -- Tempo trace retention is set via `compactor.compaction.block_retention`. Default is short (24h–48h); long retention is opt-in per cluster. - -## Grafana dashboards & datasources - -- Datasources are declared in the cluster's `grafana/custom-values.yaml` under `datasources.datasources.yaml`. Pin URLs to in-cluster Service DNS, not external endpoints. -- Dashboards bundled via `dashboardProviders` reference ConfigMaps; uniqueness of `uid` matters for panel-image links and alert dashboards. - -## Validation checklist (always) - -Before raising a PR that touches an observability chart: - -- [ ] `helm template` renders without error. -- [ ] PromQL in any new alert rule passes `promtool check rules`. -- [ ] No new high-cardinality label is added without a bound. -- [ ] Retention / PVC sizing has not been changed silently. -- [ ] If touching a `*-latest` sibling, the matching Argo Application points at it. See [argocd.md](./argocd.md). - -## Schema references - -- Per-cluster scheduling fields (must be rewritten, never copy-pasted): [../../platform/schemas/custom-values-schema.md](../../platform/schemas/custom-values-schema.md) -- Raw manifest sidecars (e.g., `external-dns-services/*.yaml` for Grafana ingress): [../../platform/schemas/raw-manifest-sidecar-schema.md](../../platform/schemas/raw-manifest-sidecar-schema.md) -- Storage backing for `vmstorage`, `loki` chunks, `tempo` blocks: [../../platform/schemas/storageclass-priorityclass-schema.md](../../platform/schemas/storageclass-priorityclass-schema.md) - -## See also - -- [helm-values.md](./helm-values.md) — values-file conventions -- [argocd.md](./argocd.md) — Argo CD interaction model -- [../escalation-matrix.md](../escalation-matrix.md) -- [../../platform/runbooks/argocd-sync-failure.md](../../platform/runbooks/argocd-sync-failure.md) diff --git a/docs/global/escalation-matrix.md b/docs/global/escalation-matrix.md deleted file mode 100644 index 7c6b63c..0000000 --- a/docs/global/escalation-matrix.md +++ /dev/null @@ -1,39 +0,0 @@ -> Per AI Blitz Plan §global. Layer: 1. Repo: devops-infra-helm-charts. - -# Escalation Matrix - -Use this table when a task crosses a Sanctity/Boundary line, when blast radius exceeds the agent's authority, or when a request belongs in a different repo. Always escalate **before** writing the change, not after. - -Primary owner: **siddharth.pal@meesho.com** -Secondary owner: **samarth.nag@meesho.com** - -Owner data is sourced from the repo-root [`repository.yaml`](../../repository.yaml). If owners change, update that file via the `registry-bootstrap` flow; do not edit `repository.yaml` by hand. See [SANCTITY_RULES.md](./SANCTITY_RULES.md) for the don't-touch list and [AGENT_BOUNDARIES.md](./AGENT_BOUNDARIES.md) for layer classification. - -## Situation → owner → channel - -| # | Situation | Who | Channel | Notes | -|---|-----------|-----|---------|-------| -| 1 | Edit suspected to silently fork an upstream chart (touching `helm-templates//templates/` or `values.yaml` of a vanilla-pulled chart) | siddharth.pal | Slack `#devops-infra` + PR review | Document the intentional fork in the chart's `README.md`. See [procedures/fork-upstream-chart.md](../platform/procedures/fork-upstream-chart.md). | -| 2 | Cross-cluster cleanup or "normalization" requested in the same PR as a feature change | siddharth.pal | Slack `#devops-infra` | Refuse in-PR; request a separate cleanup PR. See [SANCTITY_RULES.md](./SANCTITY_RULES.md) §surgical-edits. | -| 3 | TruffleHog flagged a real secret in a staged commit | siddharth.pal + secret owner (BU on-call) | Slack `#sec-incidents` (private) + revoke pipeline | NEVER bypass with `--no-verify`. Rotate the credential, then move it to External Secrets Operator. | -| 4 | Edit to `repository.yaml` requested | siddharth.pal | Slack `#devops-infra` | This file is owned by `registry-bootstrap` automation. Refuse and redirect to that pipeline. | -| 5 | Edit to an Argo CD `Application` / `ApplicationSet` manifest requested | siddharth.pal | PR against sister repo `Meesho/devops-infra-argo-config` | This repo does not own routing manifests. See [coding-guidelines/argocd.md](./coding-guidelines/argocd.md). | -| 6 | Schedule fields (`nodeSelector`, `tolerations`, `computeClass`) copy-pasted from one cluster's override to another without rewriting | siddharth.pal | PR review block | Re-derive from per-cluster node-pool topology. See [contour-nodeselector-tolerations-summary.md](../../contour-nodeselector-tolerations-summary.md) and [runbooks/pod-pending-scheduling.md](../platform/runbooks/pod-pending-scheduling.md). | -| 7 | Chart dep version bump where upstream changelog flags breaking template changes | siddharth.pal | Slack `#devops-infra` + PR review | Run `helm dependency update`, refresh `Chart.lock`, dry-run `helm template` against every consumer cluster. See [procedures/update-chart-version.md](../platform/procedures/update-chart-version.md). | -| 8 | Vault HA write-path failure (seal status, Raft peer loss, unsealed standby) | siddharth.pal + platform on-call | Slack `#sec-incidents` + PagerDuty | Production secret-store outage. Do not modify Vault overrides without on-call ack. | -| 9 | Edit to `manifests/storageclass/*.yaml` or `manifests/priorityclass//*.yaml` | siddharth.pal + cluster BU owner | PR review (two reviewers) | Cluster-wide singleton; affects every PVC / scheduling priority. See [schemas/storageclass-priorityclass-schema.md](../platform/schemas/storageclass-priorityclass-schema.md). | -| 10 | Deletion of a versioned-sibling chart (`-green`, `-vX.Y.Z`, `-latest`, `-old`) | siddharth.pal | Slack `#devops-infra` + grep sister repo | Confirm zero references in `Meesho/devops-infra-argo-config` before deletion. See [analyses/ADR-A2-blue-green-sibling-pattern.md](../../wiki/analyses/ADR-A2-blue-green-sibling-pattern.md). | - -## Refusal language - -When refusing, state: -1. The Sanctity rule or Layer constraint that's tripped. -2. Which owner to ping (from the table above). -3. The repo or pipeline the request actually belongs in (sister repo, `registry-bootstrap`, Vault on-call, etc.). - -## See also - -- [AGENT_BOUNDARIES.md](./AGENT_BOUNDARIES.md) -- [SANCTITY_RULES.md](./SANCTITY_RULES.md) -- [agent-operations-guide.md](./agent-operations-guide.md) -- Repo-root `CLAUDE.md` NEVER-DO list diff --git a/docs/platform/procedures/add-contour-route.md b/docs/platform/procedures/add-contour-route.md deleted file mode 100644 index 655606f..0000000 --- a/docs/platform/procedures/add-contour-route.md +++ /dev/null @@ -1,231 +0,0 @@ -> Per AI Blitz Plan §platform.procedures. Layer: 1. Repo: devops-infra-helm-charts. - -# Procedure — Add or modify a Contour HTTPProxy route - -> **Layer:** Layer 1 — values diff + PR. -> **Blast radius:** one Contour release × one cluster (a misrouted host can break ingress for the whole BU). -> **Approval:** consuming-app owner + cluster owner. Platform team if the change touches `contour-external*`. - -This procedure covers HTTPProxy route changes that flow through one of the cluster's Contour releases. Most clusters run **multiple Contour instances** — read the [contour-nodeselector-tolerations-summary.md](../../../contour-nodeselector-tolerations-summary.md) (repo root) authoritative scheduling matrix before editing **any** Contour values. - ---- - -## Multi-Contour topology - -| Release name | Plane | Typical purpose | -|--------------|-------|-----------------| -| `contour-external` | North-south | Public / external traffic (terminates at GCP external LB) | -| `contour-external-1` | North-south | Second external Contour (blue-green or capacity split) | -| `contour-internal-0` | East-west | Internal traffic, plane 0 | -| `contour-internal-1` | East-west | Internal traffic, plane 1 | -| `contour-internal-intra-0` | Intra-cluster | Cluster-local east-west, plane 0 | -| `contour-internal-intra-1` | Intra-cluster | Cluster-local east-west, plane 1 | - -Each is a separate Helm release, on a separate node pool, with its own `nodeSelector` / `tolerations` / `computeClass`. Picking the wrong release is the most common authoring mistake. - ---- - -## When to use - -- Adding a new HTTPProxy / route for a service. -- Changing TLS, retry, timeout, or rate-limit policy on an existing route. -- Re-pointing an HTTPProxy at a different upstream Service. -- Adding a new host to an existing route's `virtualhost.fqdn`. - -Do **not** use this procedure for: - -- Changing Contour itself (sizing, scheduling, image) — that's [modify-observability-config.md](modify-observability-config.md)-style infra editing on the Contour release. -- Bumping Contour's chart version — use [update-chart-version.md](update-chart-version.md) and consult the versioned sibling (`contour-v1.33.3`). -- Curling production hostnames to test — [SANCTITY_RULES R3](../../global/SANCTITY_RULES.md) forbids it. Test from inside the cluster with a `curl` Pod. - ---- - -## Inputs - -| Input | Example | -|-------|---------| -| Target cluster | `k8s-supply-prd-ase1` | -| Target Contour release | `contour-internal-0` | -| HTTPProxy host | `api-foo.internal.meeshogcp.in` | -| Upstream Service | `foo-svc.foo-ns:8080` | -| TLS source | `cert-manager` / `external-secret` / `none` | -| Path prefixes | `/v1`, `/health` | -| Approval ticket | CMR-… (if BU policy) | - ---- - -## Pre-conditions - -- [ ] You know which Contour release is the right one for this host (north-south = `external*`; east-west between BUs = `internal-{0,1}`; cluster-local = `internal-intra-{0,1}`). When in doubt, sample existing HTTPProxies in the same namespace. -- [ ] The upstream Service exists or will exist by the time of Sync. -- [ ] The DNS name follows the cluster's `external-dns` pattern. -- [ ] The TLS source (Secret, Issuer, etc.) exists on the cluster. - ---- - -## Steps - -### 1. Identify the right Contour release - -```bash -ls helm-overrides// | grep '^contour' -``` - -Pick the release whose plane matches the new route's traffic class. Cross-reference [contour-nodeselector-tolerations-summary.md](../../../contour-nodeselector-tolerations-summary.md). If unsure, look at existing HTTPProxies on the cluster: - -```bash -# Read existing routes already shipped via this repo -yq e '.. | select(has("httpproxies"))' \ - helm-overrides//contour-internal-0/custom-values.yaml -``` - -### 2. Decide where the HTTPProxy lives - -Two patterns exist: - -| Pattern | When | Where to author | -|---------|------|-----------------| -| **Inline in Contour values** | Routes shared by the cluster's infra layer (e.g. Grafana, Argo CD). | `helm-overrides///custom-values.yaml` under `httpproxies:` (if the chart supports it) or as a sidecar manifest in the same dir. | -| **In the consuming app's repo** | Routes for a specific service. | The service's own deployment artifacts. **Out of scope for this repo.** | - -If the route is service-owned, **redirect** to the consuming team's repo and stop. This procedure only covers infra-layer HTTPProxies. - -### 3. Author the HTTPProxy - -Skeleton (sidecar manifest pattern): - -```yaml -apiVersion: projectcontour.io/v1 -kind: HTTPProxy -metadata: - name: - namespace: -spec: - virtualhost: - fqdn: - tls: - secretName: # cert-manager-managed Secret - routes: - - conditions: - - prefix: / - services: - - name: - port: - timeoutPolicy: - response: 30s - retryPolicy: - count: 2 - retryOn: 5xx -``` - -Drop into `helm-overrides///httpproxies/.yaml` (sidecar manifest) — see [../schemas/raw-manifest-sidecar-schema.md](../schemas/raw-manifest-sidecar-schema.md). - -### 4. Lint the HTTPProxy - -```bash -# Schema check (kubectl --dry-run against the cluster's CRD) -kubectl --context= --dry-run=server -f helm-overrides///httpproxies/.yaml apply - -# Or local: validate against the projectcontour CRD schema -kubeconform -schema-location default -schema-location \ - 'https://raw.githubusercontent.com/projectcontour/contour/main/examples/contour/01-crds.yaml' \ - helm-overrides///httpproxies/.yaml -``` - -### 5. Render the chart - -```bash -helm template helm-templates/ \ - -f helm-overrides///custom-values.yaml > /tmp/contour.yaml -``` - -The render must succeed; the HTTPProxy sidecar is applied alongside, not through Helm — but rendering catches values-side errors that would block sync of the same Argo Application. - -### 6. Test from inside the cluster (post-merge, not pre-merge) - -**Do not** curl `.meeshogcp.in` from your laptop / a build agent. Run a curl Pod on-cluster: - -```bash -kubectl --context= run -it --rm curl-test \ - --image=curlimages/curl --restart=Never -- \ - curl -v -H "Host: " http://.projectcontour.svc.cluster.local -``` - -### 7. Open the PR - -```bash -git checkout -b contour/- -git add helm-overrides/// -git commit -m "contour(/): add route " -git push origin contour/- -gh pr create --base main -``` - -### PR description template - -```markdown -## Summary -Adds (or modifies) HTTPProxy `` on `` via `` for FQDN ``. - -## Why -<1-2 sentences> - -## Topology -- Cluster: `` -- Contour release: `` (plane: external / internal / intra) -- FQDN: `` -- Upstream Service: `:` in namespace `` -- TLS: cert-manager Secret `` - -## Validation -- [ ] HTTPProxy CRD schema validation passed (kubeconform / kubectl --dry-run) -- [ ] `helm template` rendered cleanly -- [ ] On-cluster curl from a curl Pod returns expected status -- [ ] DNS / external-dns plumbed (existing wildcard or new external-dns Service) - -## Approvers -- App owner: -- Cluster owner: -- Platform (if `contour-external*`): -``` - -### 8. After merge — Sync - -Open the cluster's Argo CD UI, find the Contour release's Application, **Sync**. Verify: - -```bash -kubectl --context= -n projectcontour get httpproxy -kubectl --context= -n projectcontour describe httpproxy | grep -A5 'Status:' -``` - -A `Valid: true` status means Contour accepted the route. `Valid: false` with a reason → fix the values and re-PR. - -If the HTTPProxy is `Valid` but traffic still 5xx → [../runbooks/ingress-down.md](../runbooks/ingress-down.md) §4. - ---- - -## Anti-patterns - -1. **Wrong Contour release.** A route mounted on `contour-internal-intra-0` is unreachable from outside the cluster. Cross-check the matrix. -2. **Curling the production FQDN** from a developer machine to "test." Forbidden — see [SANCTITY_RULES R3](../../global/SANCTITY_RULES.md). -3. **Hand-edited values for a single host across all Contour releases** — pick one release, justify it. -4. **`tls.passthrough` for plain HTTP services** — silent: TLS terminates upstream, doesn't. -5. **Wildcard hosts that overlap an existing HTTPProxy** — Contour status will mark one of them invalid; check before merging. -6. **Editing `helm-templates/contour*/`** to "tweak the chart." Layer 1 forbids casual chart edits — the chart is vanilla upstream. - ---- - -## Rollback - -- Revert the values PR. -- Sync the Contour Application — Argo will prune the HTTPProxy or restore the prior values. - ---- - -## Related - -- Reference (authoritative scheduling): [../../../contour-nodeselector-tolerations-summary.md](../../../contour-nodeselector-tolerations-summary.md). -- Runbook: [../runbooks/ingress-down.md](../runbooks/ingress-down.md). -- Schema: [../schemas/raw-manifest-sidecar-schema.md](../schemas/raw-manifest-sidecar-schema.md). -- Schema: [../schemas/custom-values-schema.md](../schemas/custom-values-schema.md). -- ADR: [../../../wiki/analyses/ADR-A3-per-cluster-scheduling.md](../../../wiki/analyses/ADR-A3-per-cluster-scheduling.md). diff --git a/docs/platform/procedures/blue-green-chart-migration.md b/docs/platform/procedures/blue-green-chart-migration.md deleted file mode 100644 index 8a3bbf6..0000000 --- a/docs/platform/procedures/blue-green-chart-migration.md +++ /dev/null @@ -1,206 +0,0 @@ -# Procedure — Blue-green chart migration (versioned siblings) - -> **Layer:** Layer 1 — HIGH RISK. -> **Blast radius:** the chart upgrade ships in a new sibling directory; no cluster cuts over until its `Application` is repointed. Each cluster's cutover is its own decision. -> **Approval:** platform team — multi-reviewer. - -This procedure handles the case where you can't safely bump a chart's pinned `dependencies[].version` in place — typically because of breaking template changes, immutable selector mismatches, or major-version semantics. The pattern is to keep both versions live as **sibling chart directories** until every consuming cluster has migrated. - -The repo already shows this pattern at work: - -| Original | Migration target | -|----------|------------------| -| `argo-cd` | `argo-cd-green` | -| `contour` | `contour-v1.33.3` | -| `keda` | `keda-2.17.1` | -| `opentelemetry-collector` | `opentelemetry-collector-latest` | -| `victoria-metrics-cluster` | `victoria-metrics-cluster-latest` | -| `victoria-metrics-agent` | `victoria-metrics-agent-latest` | -| `sonarqube` | `sonarqube-old` *(reverse — `sonarqube` is the new one; `-old` retained for rollback)* | - -See [ADR-A2-blue-green-sibling-pattern.md](../../../wiki/analyses/ADR-A2-blue-green-sibling-pattern.md) for the rationale. - ---- - -## When to use this procedure - -- A major-version bump with breaking template changes. -- A bump with an immutable-field change (Deployment `spec.selector`, StatefulSet `volumeClaimTemplates`). -- A migration that needs cluster-by-cluster cutover with rollback windows. - -This is **not** the right procedure for: - -- A patch / minor bump that's a clean drop-in → use [update-chart-version.md](update-chart-version.md). -- Forking templates → use [fork-upstream-chart.md](fork-upstream-chart.md). - ---- - -## Pre-conditions - -- [ ] Platform team agrees a blue-green migration is required (not just a bump). -- [ ] You've identified the variant naming convention (`-green`, `-vX.Y.Z`, `-latest`, etc.). -- [ ] You've identified every consuming cluster: `grep -rl '' helm-overrides`. -- [ ] CMR open. - ---- - -## Steps - -### 1. Create the sibling chart directory - -```bash -cp -R helm-templates/ helm-templates/- -cd helm-templates/- -``` - -Update `Chart.yaml`: - -```diff - apiVersion: v2 --name: -+name: - - … - dependencies: - - name: -- version: -+ version: - repository: -``` - -```bash -helm dependency update -``` - -### 2. Render against a representative cluster's overrides — old variant - -The point of blue-green is *no surprises*. Render the **old** chart with each consuming cluster's override and capture the output. - -```bash -for f in $(find helm-overrides -maxdepth 2 -name custom-values.yaml -path "*//*"); do - cluster=$(echo "$f" | awk -F/ '{print $2}') - helm template helm-templates/ -f "$f" > /tmp/old-${cluster}.yaml -done -``` - -### 3. Render against the same overrides — new sibling variant - -```bash -for f in $(find helm-overrides -maxdepth 2 -name custom-values.yaml -path "*//*"); do - cluster=$(echo "$f" | awk -F/ '{print $2}') - helm template - helm-templates/- -f "$f" > /tmp/new-${cluster}.yaml -done -``` - -### 4. Diff old → new per cluster - -```bash -for cluster in $(ls /tmp/old-*.yaml | sed 's:/tmp/old-::; s:.yaml::'); do - echo "=== $cluster ===" - diff /tmp/old-${cluster}.yaml /tmp/new-${cluster}.yaml | head -30 -done -``` - -For each cluster, decide: - -- Is the diff what you expected? -- Will any value need updating to make the new chart render correctly? (If yes → that's a separate per-cluster PR after the sibling lands.) -- Is the cutover safe to do without the workload owner present? (If no → schedule.) - -### 5. Open PR-1: introduce the sibling chart - -```bash -git checkout -b migrate/-to--introduce -git add helm-templates/-/ -git commit -git push origin migrate/-to--introduce -gh pr create --base main --title "migrate: introduce - sibling" -``` - -After merge, the new chart exists in the repo but **no cluster uses it yet** — the existing Argo `Application`s still point at `helm-templates/`. - -### 6. Per-cluster cutover (one PR pair per cluster) - -For each consuming cluster: - -a. **Update the cluster's `custom-values.yaml`** if the new chart needs different values. Open as a values-side PR (this repo). - -b. **Update the sister-repo `Application`** to repoint: - -```diff - spec: - source: - repoURL: https://github.com/Meesho/devops-infra-helm-charts.git - targetRevision: main -- path: helm-templates/ -+ path: helm-templates/- -``` - -c. **After both merge**, click Sync in the cluster's Argo CD UI. - -d. **Soak** — leave it for the agreed soak period (often 24–72 h) before moving to the next cluster. - -### 7. Open PR-N: retire the old sibling - -After every cluster has cut over and soaked: - -```bash -git checkout -b migrate/-retire-old -git rm -r helm-templates/ -# OR rename: git mv helm-templates/ helm-templates/-old -git commit -git push origin migrate/-retire-old -gh pr create --base main --title "migrate: retire old sibling" -``` - -PR description: - -- Confirmation every cluster has cut over (`grep -rl 'helm-templates/$' /path/to/devops-infra-argo-config` returns nothing). -- Confirmation soak period elapsed. -- Decision: delete vs rename to `-old` (kept for rollback). - ---- - -## Why two PRs at the start, then per-cluster pairs, then a final retirement - -| PR | Effect | -|----|--------| -| **PR-1: introduce sibling** | Adds the new chart. No cluster cuts over. Worst case: render errors caught before any cluster sees them. | -| **PR-2..N-1: per-cluster cutover (paired with sister repo)** | One cluster moves. Worst case: that cluster's release breaks; revert the sister-repo PR and Sync to the old chart. | -| **PR-N: retire old sibling** | Removes the old chart. Worst case: a cluster you missed becomes broken — but you grep'd, so this should be impossible. | - -Bundling any of these violates the "rollback one cluster at a time" property that's the whole point of the blue-green pattern. - ---- - -## Anti-patterns - -1. **Cutting over multiple clusters in one PR.** Bundle = no per-cluster rollback. -2. **Deleting the old sibling before every cluster has cut over.** Cluster N+1 has its `Application` pointing at a path that no longer exists; sync fails immediately. -3. **Cutting over without rendering first.** Surprises after merge. -4. **Skipping the soak period.** "It looked fine in the first 5 minutes" is not soak. -5. **Renaming the *new* sibling to drop the suffix** (e.g. `argo-cd-green` → `argo-cd`) before the old chart is retired. The path collision will break Argo CD's caching. - ---- - -## Rollback (per cluster) - -If a cluster's cutover fails: - -1. Revert the sister-repo PR for that cluster. -2. Click Sync in the cluster's Argo CD — the `Application` re-renders against `` (the old sibling). The rollback is one cluster only. -3. Investigate; iterate. - -If the new sibling has a fundamental problem affecting every cluster: - -1. Open PR-X: revert PR-1 (delete the sibling). -2. Any cluster that was already cutover gets reverted via its own per-cluster sister-repo revert. -3. Schedule a postmortem before re-attempting. - ---- - -## Related - -- Procedure: [update-chart-version.md](update-chart-version.md). -- Procedure: [fork-upstream-chart.md](fork-upstream-chart.md). -- ADR: [ADR-A2-blue-green-sibling-pattern.md](../../../wiki/analyses/ADR-A2-blue-green-sibling-pattern.md). -- [SANCTITY_RULES R8](../../global/SANCTITY_RULES.md). diff --git a/docs/platform/procedures/deboard-app.md b/docs/platform/procedures/deboard-app.md deleted file mode 100644 index 4d2c3a4..0000000 --- a/docs/platform/procedures/deboard-app.md +++ /dev/null @@ -1,119 +0,0 @@ -# Procedure — Deboard a retired app from a cluster - -> **Layer:** Layer 1 — HIGH RISK. -> **Blast radius:** the Helm release is removed from the cluster. If still in use, that's an outage. -> **Approval:** app owner + cluster owner. CMR mandatory. - -This procedure removes an app's `helm-overrides///` directory. It does **not** delete the workload directly — but the matching sister-repo `Application` PR (which must be paired) tells Argo CD to stop tracking the release. - ---- - -## Pre-conditions (block deboarding if any are FALSE) - -- [ ] App is genuinely retired on this cluster — confirmed by app owner. -- [ ] No traffic / scrape job / dependency is still hitting it. -- [ ] All upstream consumers (alerts, dashboards, log collectors) have been migrated or notified. -- [ ] You have inventoried every cluster that runs the app and decided whether this is a single-cluster or fleet-wide deboard. -- [ ] CMR approved. - -```bash -# Where does this app exist today? -find helm-overrides -maxdepth 2 -type d -name '' -``` - ---- - -## Steps - -### 1. Decide the scope - -| Scope | Then | -|-------|------| -| Single cluster | Remove `helm-overrides///` only. Leave other clusters running. | -| Fleet-wide | Multiple PRs — one cluster per PR. Don't bundle. | -| App is being replaced (e.g. `victoria-metrics-cluster` → `victoria-metrics-cluster-latest`) | This is a **migration**, not a deboard — use [blue-green-chart-migration.md](blue-green-chart-migration.md). | - -### 2. Open the sister-repo `Application` removal PR FIRST - -In `github.com/Meesho/devops-infra-argo-config`: - -- Remove (or scope out of) the `Application` / `ApplicationSet` entry that targets this cluster × app. -- Merge. -- The cluster's Argo CD will mark the `Application` for removal on next reconcile. - -This step **must precede** the values-side removal. If you delete the values-side first, the `Application` will fail to render and the workload may go into an `Errored` state on the cluster. - -### 3. Manually clean up the workload (if `automated.prune` was not set) - -For most infra apps, the `Application` does not have `automated.prune: true` — so removing the `Application` does not delete the workload. Manually: - -```bash -kubectl --context= delete / -n -# OR, if the whole namespace is dedicated to this release: -kubectl --context= delete namespace -``` - -Coordinate with the app owner — wholesale namespace deletion is irreversible. - -### 4. Open the values-side removal PR - -```bash -git checkout -b deboard/-from- -git rm -r helm-overrides/// -git commit -git push origin deboard/-from- -gh pr create --base main --title "deboard: from " -``` - -PR description: - -- Procedure followed: this file. -- Confirmation pre-conditions are TRUE. -- Sister-repo PR (already merged). -- CMR ticket reference. -- Confirmation the workload was manually cleaned up (or scheduled). - -### 5. (Optional) If this was the last cluster running the app - -If `find helm-overrides -maxdepth 2 -type d -name ''` now returns nothing, consider: - -- **Should `helm-templates//` also be removed?** Probably not — keeping the chart cached lets a future re-onboarding be cheap. But if it's a stale chart with security advisories you don't want to maintain, schedule a separate retirement PR for it (with platform-team review). - ---- - -## "What if the app might come back?" - -If retirement is provisional: - -- **Do not** scale the workload to zero replicas via values "to deactivate it." Either it's running or it's not. Half-states are operational debt. -- **Do** keep the values directory in place but document a 30-day decision deadline in a TODO comment. If the deadline passes without a re-decision, deboard for real. - ---- - -## Anti-patterns - -1. **Deleting the values-side first** before the sister-repo `Application` is removed. The `Application` errors on next reconcile. -2. **Bulk-deleting multiple clusters in one PR.** One cluster per PR. Rollback granularity. -3. **Forgetting to clean up the workload manually.** The Helm release lingers on the cluster after the `Application` is removed (because most infra apps lack `automated.prune`). -4. **Forgetting `external-secrets` cleanup.** If the app referenced an `ExternalSecret`, the corresponding `ExternalSecret` resource (in the cluster's `external-secrets/` directory) often outlives the app. Either repurpose it or delete it in the same PR. -5. **Forgetting alert / dashboard cleanup.** Alerts firing on a workload that no longer exists cause noise; dashboards showing nothing cause confusion. - ---- - -## Rollback - -If you deboarded by mistake: - -1. Revert the values-side PR (`git revert `). -2. Revert the sister-repo PR. -3. Click Sync in the cluster's Argo CD UI. -4. The `Application` is recreated and renders the chart. -5. **However**, if you also manually deleted the workload (step 3), the Sync recreates it from scratch — make sure the underlying chart and values are still intact and any data PVCs were retained (see `manifests/storageclass/pd-standard-retain-dr.yaml`). - ---- - -## Related - -- Procedure: [onboard-app-to-cluster.md](onboard-app-to-cluster.md) — the inverse. -- Procedure: [blue-green-chart-migration.md](blue-green-chart-migration.md) — if "deboard" is actually a migration. -- ADR: [ADR-A5-manual-sync-default-for-infra.md](../../../wiki/analyses/ADR-A5-manual-sync-default-for-infra.md) — why the workload doesn't auto-delete. diff --git a/docs/platform/procedures/fork-upstream-chart.md b/docs/platform/procedures/fork-upstream-chart.md deleted file mode 100644 index b98c405..0000000 --- a/docs/platform/procedures/fork-upstream-chart.md +++ /dev/null @@ -1,158 +0,0 @@ -# Procedure — Intentionally fork an upstream chart - -> **Layer:** Layer 1 — HIGH RISK. -> **Blast radius:** every cluster that consumes this chart will render against the forked templates on next sync. -> **Approval:** platform team — multi-reviewer. - -Most charts in `helm-templates/` are **vanilla upstream**, pulled via `helm pull`. The `Chart.yaml` is a thin wrapper declaring the upstream chart as a dependency; the actual templates come from the subchart in `charts/`. Editing a `templates/*.yaml` *under the wrapper* silently forks the chart, and the fork is clobbered the next time someone runs `helm dependency update`. - -This procedure is the supported path when a fork is **intentional**. - ---- - -## When to use this procedure - -- Upstream chart lacks a feature you need (e.g. a values key that doesn't exist). -- Upstream behaviour conflicts with Meesho's environment (e.g. probe path doesn't work behind Contour). -- Upstream chart has a bug awaiting an upstream fix. - -This is **not** the right procedure for: - -- Adding a values knob — file a PR upstream, or override behaviour through existing knobs. -- Pinning to an old version — use [update-chart-version.md](update-chart-version.md) (or just don't bump). -- Style preferences — leave the upstream chart alone. - ---- - -## Pre-conditions - -- [ ] You've confirmed the desired behaviour cannot be achieved via the chart's existing values. -- [ ] You've checked whether an upstream PR / issue already covers this. -- [ ] Platform team agrees the fork is justified. -- [ ] The fork's rationale will be documented in the chart's `README.md`. - ---- - -## Steps - -### 1. Vendor the chart's templates - -If the chart is currently a thin wrapper (templates come from a subchart in `charts/`), you must first promote the subchart's templates into the wrapper. - -```bash -cd helm-templates/ -ls charts/ # find the subchart .tgz -helm dependency update # ensure it's resolved - -# Extract templates from the subchart .tgz -tar -xzf charts/-.tgz -C /tmp/ -cp -R /tmp//templates ./templates -cp -R /tmp//values.yaml ./values.yaml.upstream -``` - -Now the wrapper has its own `templates/` — Helm will use those instead of the subchart's. - -### 2. Edit `Chart.yaml` to reflect the fork - -Remove the dependency (since the templates are now local), and bump `version:` (the chart's own version, not the subchart's): - -```diff - apiVersion: v2 - name: --version: 0.1.0 -+version: 0.1.0+fork.1 - description: — Meesho-forked from upstream --dependencies: -- - name: -- version: -- repository: -``` - -### 3. Make the fork edits - -Edit `templates/*.yaml` or `values.yaml` to apply the fix. Keep the diff minimal — every line away from upstream is technical debt. - -### 4. Document the fork - -Edit `helm-templates//README.md` (create if missing). Use this template: - -```markdown -# - -**Forked from upstream ** at . - -## Why - - - -## What's changed - -- `templates/.yaml` — -- `values.yaml` — - -## Upstream tracking - -- Upstream PR: -- Upstream issue: -- When upstream merges: revert this fork via [unfork procedure]. -``` - -### 5. Render and diff - -```bash -sibling=$(find helm-overrides -maxdepth 2 -type d -name '' | head -1) -helm template helm-templates/ -f "$sibling/custom-values.yaml" | head -120 - -# Spot-check helm diff against a live cluster -helm diff upgrade helm-templates/ \ - -f helm-overrides///custom-values.yaml --kube-context= -``` - -### 6. Commit and open the PR - -```bash -git checkout -b fork/- -git add helm-templates// -git commit -git push origin fork/- -gh pr create --base main --title "fork: " -``` - -PR description must include: - -- Procedure followed: this file. -- Why the fork is necessary (link to upstream issue/PR if any). -- What templates / values are changed and why. -- Platform team approver(s). -- Plan for unforking when upstream lands the fix. - ---- - -## Anti-patterns - -1. **Editing `templates/` without first vendoring** (templates from a subchart). Your edits live in `helm-templates//templates/` but Helm renders from `charts//templates/` — the edits do nothing, then get clobbered. -2. **Forking a chart and not documenting why.** Six months later nobody remembers; the fork looks like accidental drift. -3. **Forking instead of overriding via values.** Always check the chart's existing values surface first. -4. **Forking and then bumping the upstream version.** A bump runs `helm dependency update`, which can clobber the fork. The fork must be re-applied or the bump must explicitly re-vendor. -5. **Sweeping cleanup of upstream code** alongside the fork edit. The diff should be exactly the change you intended; everything else is upstream. - ---- - -## Rollback / unforking - -When upstream lands the fix and you want to return to vanilla: - -1. Restore `Chart.yaml` to declare the upstream as a dependency (with the new upstream version that includes the fix). -2. Delete the local `templates/` and `values.yaml` (or rename them as `.upstream` for reference). -3. Run `helm dependency update`. -4. Update `helm-templates//README.md` to remove the "forked" status. -5. Render against a sibling override and confirm the resulting manifests match what the fork was producing. - ---- - -## Related - -- Procedure: [update-chart-version.md](update-chart-version.md). -- ADR: [ADR-A1-cache-vs-upstream-charts.md](../../../wiki/analyses/ADR-A1-cache-vs-upstream-charts.md). -- [SANCTITY_RULES R7](../../global/SANCTITY_RULES.md). diff --git a/docs/platform/procedures/modify-alert-rules.md b/docs/platform/procedures/modify-alert-rules.md deleted file mode 100644 index 017d174..0000000 --- a/docs/platform/procedures/modify-alert-rules.md +++ /dev/null @@ -1,227 +0,0 @@ -> Per AI Blitz Plan §platform.procedures. Layer: 1. Repo: devops-infra-helm-charts. - -# Procedure — Modify Prometheus / VictoriaMetrics alert rules - -> **Layer:** Layer 1 — values diff + PR. -> **Blast radius:** PagerDuty / on-call paging across the cluster's tenants. A wrong threshold pages everyone; a missing rule masks an outage. -> **Approval:** observability owner + on-call lead for the affected severity. - -This procedure is the narrow alert-rule slice of [modify-observability-config.md](modify-observability-config.md). Use this when the entire intent of the PR is alert-rule editing — adding, removing, retuning thresholds, changing `for:` windows, or rewriting `expr:` PromQL. - ---- - -## Where alert rules live - -| File | Engine | Notes | -|------|--------|-------| -| `helm-overrides//vmalert/custom-values.yaml` | vmalert | Most clusters use vmalert against VM cluster as the primary alerting engine. Rules typically under `config.rules` or rendered as a `VMRule` CRD. | -| `helm-overrides//victoria-metrics-cluster/custom-values.yaml` | vmalert (bundled) / VM ruler | Some clusters carry rules in the cluster chart's `vmalert.config` block. | -| `helm-overrides//prometheus/custom-values.yaml` | Prometheus / `kube-prometheus-stack` | Rules under `additionalPrometheusRulesMap` or `serverFiles.alerting_rules.yml`. | -| `helm-overrides//kube-prometheus-stack/custom-values.yaml` | Prometheus | Same as above; older clusters use this chart name. | -| `helm-overrides//mimir/custom-values.yaml` | Mimir ruler | Rules go through the ruler API; YAML in `runtimeConfig` or via a sidecar `ConfigMap`. | - -Confirm which cluster runs which engine before editing — sample existing rules in the override. - ---- - -## When to use - -- Adding/removing one or more alert rules in the cluster's primary alerting engine. -- Tuning `expr:` PromQL or `for:` windows on existing rules. -- Re-routing alerts (changing `severity`, `team`, or other routing labels). -- Adding/removing recording rules for downstream alert efficiency. - -Do **not** use this procedure for: - -- Touching scrape configs, retention, datasources, or sizing in the same PR — split it. See [modify-observability-config.md](modify-observability-config.md). -- Bumping the alerting chart's version — see [update-chart-version.md](update-chart-version.md). -- Writing app-specific alert rules that belong in the consuming service's own values — those go in the service's repo. - ---- - -## Inputs - -| Input | Example | -|-------|---------| -| Target cluster | `k8s-supply-prd-ase1` | -| Target chart | `vmalert` / `prometheus` / `kube-prometheus-stack` / `mimir` | -| Rule name(s) added/changed | `HighErrorRate`, `KafkaConsumerLag` | -| `expr:` PromQL | `sum(rate(http_requests_total{status=~"5.."}[5m])) by (service) > 0.05` | -| `for:` window | `5m` | -| Severity / routing labels | `severity: critical`, `team: supply-platform` | -| Approval ticket | CMR-… | - ---- - -## Pre-conditions - -- [ ] You have `promtool` and/or `vmalert` available locally. -- [ ] You know the cluster's Alertmanager / vmalert notifier routing (which `severity` / `team` label routes where). -- [ ] You have run the new PromQL expression against the cluster's VM/Prometheus to confirm it returns sensible values *before* writing the rule — open Grafana Explore. -- [ ] You have on-call sign-off if this rule pages on-call. - ---- - -## Steps - -### 1. Locate the rules block - -```bash -yq e '.config.rules // .additionalPrometheusRulesMap // .serverFiles' \ - helm-overrides///custom-values.yaml -``` - -Identify the YAML path the engine expects: - -- **vmalert**: `config.groups[].rules[]`. -- **kube-prometheus-stack**: `additionalPrometheusRulesMap..groups[].rules[]`. -- **prometheus**: `serverFiles.alerting_rules.yml.groups[].rules[]`. -- **mimir ruler**: per-tenant config — confirm the cluster's tenancy setup. - -### 2. Author the rule - -```yaml -- alert: # PascalCase, no spaces; surfaces in pages and dashboards - expr: | - - for: # e.g. 5m, 10m - labels: - severity: - team: # routes to the right PagerDuty service - annotations: - summary: "" - description: "" - runbook_url: -``` - -#### Threshold discipline - -- Express thresholds as `rate(...) > 0.05` rather than `> 5%`. Ensure units in the expression match the units in `summary` / `description`. -- Avoid `for:` windows shorter than the scrape interval × 2. Sub-`1m` windows are flap factories. -- Avoid alerting on absolute counts (`sum(http_requests_total) > 1000`); prefer rates / ratios. -- For SLO-style alerts, use multi-window multi-burn-rate (Google SRE workbook). Single-window threshold alerts are a known anti-pattern. - -#### Routing implications - -Changing `severity` from `warning` to `critical` (or vice versa) **silently re-routes the page** — the rule may now wake on-call where it previously emailed (or vice versa). Same for `team:` — a typo here drops alerts into the wrong PagerDuty service. Cross-check the cluster's Alertmanager / vmalert routing config: - -```bash -yq e '.config.route // .alertmanagerConfig' \ - helm-overrides///custom-values.yaml -``` - -### 3. Render the chart - -```bash -helm template helm-templates/ \ - -f helm-overrides///custom-values.yaml > /tmp/rendered.yaml -``` - -### 4. Validate PromQL - -For Prometheus / `kube-prometheus-stack`: - -```bash -yq e 'select(.kind == "PrometheusRule")' /tmp/rendered.yaml > /tmp/rules.yaml -promtool check rules /tmp/rules.yaml -``` - -For vmalert: - -```bash -yq e 'select(.kind == "VMRule" or .kind == "ConfigMap")' /tmp/rendered.yaml > /tmp/vmrules.yaml -vmalert -dryRun -rule=/tmp/vmrules.yaml -``` - -For Mimir ruler — `mimirtool rules check` against the rendered ruler config. - -If validation fails: fix the PromQL. **Do not** merge a rule that fails parse. - -### 5. Sanity-check expressions in Grafana Explore - -In a non-production manner: open Grafana on the cluster, paste the `expr:` into Explore, run over the last 6h. Confirm: - -- The series returned makes sense for the rule. -- The threshold isn't always-firing or never-firing in normal operation. -- Empty result is allowed (`absent_over_time(...)` style). - -### 6. Open the PR - -```bash -git checkout -b alerts/- -git add helm-overrides///custom-values.yaml -git commit -m "alerts(/): " -git push origin alerts/- -gh pr create --base main -``` - -### PR description template - -```markdown -## Summary - alert rule(s) on `` × ``. - -## Rules touched -| Rule | Change | New expr | New `for:` | Severity | Team | -|------|--------|----------|-----------|----------|------| -| `` | added | `` | `5m` | `critical` | `supply-platform` | - -## Validation -- [ ] `promtool check rules` / `vmalert -dryRun` passed -- [ ] PromQL sanity-checked in Grafana Explore over last 6h -- [ ] Threshold appropriate (no flap risk, no always-firing) -- [ ] Routing labels (`severity`, `team`) verified against Alertmanager/vmalert routes -- [ ] Runbook URL added (`annotations.runbook_url`) - -## Approvers -- Observability owner: -- On-call lead (severity=): -``` - -### 7. After merge — Sync and observe - -Sync the Argo Application. Watch: - -```bash -# vmalert -kubectl --context= -n monitoring logs deploy/vmalert | grep -i "\|error" - -# kube-prometheus-stack -kubectl --context= -n monitoring logs prometheus-k8s-0 | grep -i "\|error" -``` - -Open the alerting engine's UI (vmalert UI / Prometheus `/alerts`) to confirm the rule loaded. - -### Watch for false positives - -Stay on the alert for at least one full `for:` window after sync. If the rule pages immediately and was not expected to → revert. - ---- - -## Anti-patterns - -1. **Sub-1m `for:` windows** — flap-prone. -2. **Alerting on absolute counts** instead of rates/ratios. -3. **`severity` flips without on-call sign-off** — silent re-route. -4. **Missing `runbook_url`** — pages without remediation steps waste on-call cycles. -5. **PromQL that uses `{job=~"foo.*"}` regex without anchors** — slow query, sometimes false matches. -6. **Inlining the rule body in the chart `templates/`** to "make it sticky" — that's a chart fork; see NEVER-DO. -7. **Adding the rule on every cluster in one PR** — surgical only. - ---- - -## Rollback - -- Revert the PR. Sync. -- vmalert/Prometheus reload picks up the previous ruleset within seconds. -- Any in-flight alerts resolve on next evaluation. - ---- - -## Related - -- Coding guideline: [../../global/coding-guidelines/observability.md](../../global/coding-guidelines/observability.md). -- Procedure: [modify-observability-config.md](modify-observability-config.md) — broader observability edits. -- Runbook: [../runbooks/metrics-gap.md](../runbooks/metrics-gap.md) — when an alert is silent because the metric is gone. -- Schema: [../schemas/custom-values-schema.md](../schemas/custom-values-schema.md). -- Escalation: [../../global/escalation-matrix.md](../../global/escalation-matrix.md). diff --git a/docs/platform/procedures/modify-observability-config.md b/docs/platform/procedures/modify-observability-config.md deleted file mode 100644 index bd9625b..0000000 --- a/docs/platform/procedures/modify-observability-config.md +++ /dev/null @@ -1,210 +0,0 @@ -> Per AI Blitz Plan §platform.procedures. Layer: 1. Repo: devops-infra-helm-charts. - -# Procedure — Modify observability stack config - -> **Layer:** Layer 1 — values diff + PR. -> **Blast radius:** one chart × one cluster (a metrics gap, retention change, or alert-rule edit can ripple to paging and dashboards across the BU). -> **Approval:** observability owner (`siddharth.pal@meesho.com`) + cluster owner. - -This procedure covers edits to the cluster-by-cluster observability override files: - -| Override path | What it controls | -|---------------|------------------| -| `helm-overrides//victoria-metrics-agent/custom-values.yaml` | Scrape targets, relabel rules, remote-write tenancy. | -| `helm-overrides//victoria-metrics-cluster/custom-values.yaml` | vmstorage retention, vmselect/vminsert sizing, ruler config. | -| `helm-overrides//mimir/custom-values.yaml` (also `mimir-distributed`) | Mimir distributor/ingester/ruler config and tenancy. | -| `helm-overrides//loki/custom-values.yaml` (also `loki-distributed`) | Loki retention, ingestion limits, multi-tenant config. | -| `helm-overrides//tempo/custom-values.yaml` (also `tempo-distributed`) | Tempo block retention, distributor sizing. | -| `helm-overrides//grafana/custom-values.yaml` | Datasources, dashboard providers, plugins. | -| `helm-overrides//vmalert/custom-values.yaml` | vmalert alerting rules and notifier config. | -| `helm-overrides//prometheus/custom-values.yaml` (or `kube-prometheus-stack`) | Prometheus alert rules, scrape config, retention. | - -All of these are **versioned-sibling-aware** — confirm whether the cluster's Argo Application points at `victoria-metrics-cluster` or `victoria-metrics-cluster-latest`, etc., before editing. See [../../global/coding-guidelines/observability.md](../../global/coding-guidelines/observability.md) for stack conventions. - ---- - -## When to use - -- Adding/removing scrape targets, relabel rules, recording rules. -- Changing retention (`vmstorage.retentionPeriod`, Loki `retention_period`, Tempo `compaction.block_retention`). -- Adding/removing/tuning alert rules in `vmalert` or `kube-prometheus-stack`. -- Adding/removing Grafana datasources, dashboards, plugins. -- Tuning ingester/distributor sizing for Mimir/Loki/Tempo. - -Do **not** use this procedure for: - -- **Bumping the chart version** of an observability tool — use [update-chart-version.md](update-chart-version.md). -- **Onboarding a brand-new observability tool** to a cluster — use [onboard-app-to-cluster.md](onboard-app-to-cluster.md). -- **A blue-green migration** between sibling charts (`-latest`) — use [blue-green-chart-migration.md](blue-green-chart-migration.md). - ---- - -## Inputs - -| Input | Example | -|-------|---------| -| Target cluster | `k8s-supply-prd-ase1` | -| Target chart | `victoria-metrics-cluster` | -| Versioned-sibling target (if applicable) | `victoria-metrics-cluster-latest` | -| Change kind | `retention bump`, `new scrape target`, `new alert rule`, `dashboard add` | -| Promql expression(s) touched (if any) | `sum(rate(...)) by (job) > 0.05` | -| Approval ticket | CMR-1234 (if BU policy requires) | - ---- - -## Pre-conditions - -- [ ] The cluster directory and chart override exist. -- [ ] The cluster's Argo `Application` points at the chart you're editing (and not its sibling). Check `github.com/Meesho/devops-infra-argo-config`. -- [ ] You have the previous PR diff for context (most observability edits are touching a known knob). -- [ ] You have access to PromQL/promtool (or vmalert binary) locally for rule validation. - ---- - -## Steps - -### 1. Confirm which sibling the cluster runs - -```bash -# In the sister repo -gh search code --repo Meesho/devops-infra-argo-config "" -- path:**/* -``` - -Find the `Application` whose `spec.source.path` points at this repo. Note whether it's `victoria-metrics-cluster` or `victoria-metrics-cluster-latest`. **Editing the wrong one is silent** — the file diff merges, but no cluster picks it up. - -### 2. Read the current values - -```bash -yq e '.' helm-overrides///custom-values.yaml | less -``` - -Note current retention, scrape targets, and any inline alert rules. - -### 3. Author the change - -Edit `helm-overrides///custom-values.yaml`. Surgical — only the keys the task requires. Preserve YAML key order; preserve comments. - -#### Cardinality discipline - -If the change adds a label, scrape target, or `relabel_configs` rule, ask: can the new label exceed ~few-hundred distinct values? If yes, drop or aggregate. See [observability.md §Cardinality discipline](../../global/coding-guidelines/observability.md). - -#### Retention changes - -Increasing `vmstorage.retentionPeriod`, Loki `retention_period`, or Tempo `block_retention` grows the backing PVC. Confirm `persistence.size` (or `vmstorage.persistentVolume.size`) has been bumped to match — otherwise vmstorage runs out of disk silently. - -#### Alert-rule edits - -If the rule's `for:` window or `severity` label changes, the PagerDuty routing may swap. Cross-check the cluster's Alertmanager / vmalert notifier config before merging. - -### 4. Render the chart locally - -```bash -helm template helm-templates/ \ - -f helm-overrides///custom-values.yaml > /tmp/rendered.yaml -``` - -The render must succeed. If it errors, fix the values before continuing. - -### 5. Validate any PromQL touched - -For Prometheus/`kube-prometheus-stack`: - -```bash -# Extract PrometheusRule objects from rendered output -yq e 'select(.kind == "PrometheusRule")' /tmp/rendered.yaml > /tmp/rules.yaml - -# Validate -promtool check rules /tmp/rules.yaml -``` - -For vmalert: - -```bash -# Extract VMRule (or ConfigMap with rules) -yq e 'select(.kind == "VMRule" or .kind == "ConfigMap")' /tmp/rendered.yaml > /tmp/vmrules.yaml - -# vmalert dry-run -vmalert -dryRun -rule=/tmp/vmrules.yaml -``` - -If `promtool` / `vmalert` is unavailable locally, surface the rule expression in the PR description and request reviewer to validate. - -### 6. (Grafana) Validate datasource URL is in-cluster - -Datasources should point at in-cluster Service DNS (e.g. `http://victoria-metrics-cluster-vmselect:8481`), not external endpoints. **Never** pin a Grafana datasource at `*.meeshogcp.in`, `prd.meesho.int`, or any production hostname — that violates the NEVER-DO list and routes through external networking unnecessarily. - -### 7. Open the PR - -```bash -git checkout -b obs/-- -git add helm-overrides///custom-values.yaml -git commit -m "obs(/): " -git push origin obs/-- -gh pr create --base main -``` - -### PR description template - -```markdown -## Summary - - -## Cluster × chart -- Cluster: `` -- Chart: `` (sibling: ``) -- File: `helm-overrides///custom-values.yaml` - -## Validation -- [ ] `helm template` renders cleanly -- [ ] `promtool check rules` / `vmalert -dryRun` passed (PromQL expression: ``) -- [ ] Cardinality bounded (no unbounded label introduced) -- [ ] Retention/PVC headroom confirmed (if retention changed) -- [ ] PagerDuty routing unchanged (if alert rule changed) - -## Approvers -- Observability owner: -- Cluster owner: - -## Sister repo -- N/A (no Application change required) -``` - -### 8. After merge - -Argo CD on the target cluster reconciles. Most observability charts use **manual sync** (see [observability.md](../../global/coding-guidelines/observability.md)) — open the cluster's Argo CD UI, find the Application, click **Sync**. Watch: - -```bash -kubectl --context= -n get pods -w -kubectl --context= -n logs -pod | tail -50 -``` - -If a metric goes missing post-sync → [../runbooks/metrics-gap.md](../runbooks/metrics-gap.md). - ---- - -## Anti-patterns - -1. **Editing the wrong sibling.** The Argo Application points at `*-latest`; you edit the stable chart's values. Diff merges, nothing applies. -2. **Adding `pod_name` / `request_id` / `user_id` as a label** without aggregation — explodes cardinality. -3. **Bumping retention without bumping PVC.** vmstorage runs out of disk; ingest fails silently. -4. **Silent PagerDuty re-route.** Changing `severity:` from `warning` to `critical` (or vice versa) without coordinating on-call. -5. **Inlining production hostnames** as Grafana datasource URLs — see [SANCTITY_RULES.md](../../global/SANCTITY_RULES.md) R3. -6. **Cross-cluster normalising** — touching every cluster's `custom-values.yaml` in one PR. Surgical only; one cluster per PR. - ---- - -## Rollback - -- Revert the values PR. Argo CD will re-render with the previous values; click Sync. -- For retention shrinks: data older than the new retention is dropped on next compaction. Reverting restores the *config* but not the *data*. - ---- - -## Related - -- Coding guideline: [../../global/coding-guidelines/observability.md](../../global/coding-guidelines/observability.md). -- Runbook: [../runbooks/metrics-gap.md](../runbooks/metrics-gap.md). -- Procedure: [modify-alert-rules.md](modify-alert-rules.md) — alert-only edits. -- Schema: [../schemas/custom-values-schema.md](../schemas/custom-values-schema.md). -- ADR: [../../../wiki/analyses/ADR-A2-blue-green-sibling-pattern.md](../../../wiki/analyses/ADR-A2-blue-green-sibling-pattern.md). -- Wiki: [../../../wiki/entities/DevOps%20Infra%20Helm%20Charts.md](../../../wiki/entities/DevOps%20Infra%20Helm%20Charts.md). diff --git a/docs/platform/procedures/onboard-app-to-cluster.md b/docs/platform/procedures/onboard-app-to-cluster.md deleted file mode 100644 index ae1abf8..0000000 --- a/docs/platform/procedures/onboard-app-to-cluster.md +++ /dev/null @@ -1,204 +0,0 @@ -# Procedure — Onboard a new app to an existing cluster - -> **Layer:** Layer 1 — Agent-Writable. -> **Blast radius:** one new Helm release on one cluster. -> **Approval:** app owner + cluster owner. - -This procedure adds a new infrastructure-tooling Helm release to a cluster that already exists in `helm-overrides/`. It covers the slice owned by `devops-infra-helm-charts`. The matching Argo `Application` lives in `github.com/Meesho/devops-infra-argo-config` and must be paired. - ---- - -## Inputs - -| Input | Example | -|-------|---------| -| Chart name (must exist in `helm-templates/`) | `kube-state-metrics` | -| Target cluster directory | `k8s-supply-prd-ase1` | -| Release name | `kube-state-metrics` (matches chart name; may be different for variants) | -| Workload namespace | `monitoring` | -| Image tag | `v2.10.1` | -| Sized resources (CPU/memory requests + limits) | `250m / 512Mi` | -| Node-pool key (per cluster) | `dedicated: monitoring` | -| Whether this needs a sidecar `external-dns` Service | yes / no | - ---- - -## Pre-conditions - -- [ ] The chart exists in `helm-templates//` with a current `Chart.yaml`. -- [ ] The cluster directory exists in `helm-overrides//`. -- [ ] The chart is appropriate for an *infra* release (services don't go here — they live in `devops-argo-config`). -- [ ] The matching sister-repo `Application` PR is drafted (or will be drafted in parallel). -- [ ] CMR ticket open if required by BU policy. - ---- - -## Steps - -### 1. Verify the chart renders with a sibling cluster's values - -Pick a cluster that already runs this chart and use its values as a starting point: - -```bash -sibling=$(find helm-overrides -maxdepth 2 -type d -name '' | head -1) -helm template helm-templates/ -f "$sibling/custom-values.yaml" | head -60 -``` - -If the render errors → the chart's dependencies may be unresolved. Run `helm dependency update helm-templates/` first. - -### 2. Identify the cluster's scheduling profile - -```bash -# Standard GKE: uses 'dedicated:' keys -grep -rh 'dedicated:' helm-overrides//*/custom-values.yaml | sort -u - -# GKE Autopilot: uses 'cloud.google.com/compute-class' keys -grep -rh 'cloud.google.com/compute-class' helm-overrides//*/custom-values.yaml | sort -u -``` - -For Contour, cross-reference [contour-nodeselector-tolerations-summary.md](../../../contour-nodeselector-tolerations-summary.md). For others, copy from a sibling app on the **same** cluster — never from the same app on a different cluster ([SANCTITY_RULES R5](../../global/SANCTITY_RULES.md)). - -### 3. Create the directory and `custom-values.yaml` - -```bash -mkdir -p helm-overrides// -$EDITOR helm-overrides///custom-values.yaml -``` - -Author from scratch using [custom-values-schema.md](../schemas/custom-values-schema.md) and the cluster's scheduling profile from step 2. Do **not** copy a sibling cluster's values verbatim. - -Skeleton: - -```yaml -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/ - tag: - -replicaCount: - -resources: - requests: {cpu: , memory: } - limits: {cpu: , memory: } - -nodeSelector: - : -tolerations: - - {key: , value: , effect: NoSchedule} -``` - -### 4. (If needed) Add sidecar raw manifests - -If the app needs sidecar resources (e.g. `external-dns` `Service`, `ComputeClass`, `ExternalSecret`), drop them in the same directory under a subfolder: - -``` -helm-overrides/// - custom-values.yaml - external-dns-services/.yaml - computeclass/-cc.yaml -``` - -See [raw-manifest-sidecar-schema.md](../schemas/raw-manifest-sidecar-schema.md). - -### 5. Validate locally - -```bash -yamllint helm-overrides///custom-values.yaml - -# Render -helm template helm-templates/ \ - -f helm-overrides///custom-values.yaml | head -60 - -# Optional: dry-run diff against the live cluster (requires kubectl context + helm-diff plugin) -helm diff upgrade helm-templates/ \ - -f helm-overrides///custom-values.yaml \ - --kube-context= -``` - -### 6. Open the sister-repo PR - -In `github.com/Meesho/devops-infra-argo-config`, draft an `Application` (or add to an existing `ApplicationSet`): - -```yaml -apiVersion: argoproj.io/v1alpha1 -kind: Application -metadata: - name: - -spec: - destination: - name: - namespace: - source: - repoURL: https://github.com/Meesho/devops-infra-helm-charts.git - targetRevision: main - path: helm-overrides// - helm: - valueFiles: [custom-values.yaml] - syncPolicy: - syncOptions: [CreateNamespace=true] - # Most infra apps DO NOT use automated sync — see ADR-A5 -``` - -### 7. Open the values-side PR (this repo) - -```bash -git checkout -b onboard/-on- -git add helm-overrides/// -git commit -# pre-commit hook runs TruffleHog -git push origin onboard/-on- -gh pr create --base main --title "Onboard to " -``` - -PR description: - -- Procedure followed: this file. -- App owner approver tag. -- Cluster owner approver tag. -- Sister-repo PR link (`devops-infra-argo-config#`). -- CMR ticket reference (if applicable). -- Confirmation that the chart renders cleanly and `helm diff` (if run) showed only additions. - -### 8. After both merge — sync in Argo CD - -`devops-infra-argo-config`'s reconciler will create the `Application` resource on the cluster's Argo CD. Most infra apps are **manual sync** ([ADR-A5](../../../wiki/analyses/ADR-A5-manual-sync-default-for-infra.md)), so the workload deploy is a separate step: - -1. Open the cluster's Argo CD UI. -2. Search for the new `Application`. -3. Verify the manifest renders (Diff view shows the chart's resources). -4. Click **Sync**. -5. Watch the rollout: `kubectl --context= get pods -n -w`. - ---- - -## Anti-patterns - -1. **Copying the entire `helm-overrides///` directory** verbatim. Per-cluster scheduling differs. -2. **Bundling onboarding with a chart-version bump.** Two separate PRs. -3. **Skipping the sister-repo PR.** Without an `Application`, the values do nothing. -4. **Setting `automated.{prune,selfHeal}: true`** in the sister-repo `Application` "to make life easier." Manual sync is the default safety property. -5. **Inlining secrets** in `custom-values.yaml`. Use `ExternalSecret`. - ---- - -## Rollback - -If the merge causes a problem before Sync: - -- Revert the values-side PR (and the sister-repo PR). -- Argo CD will prune the `Application` resource on next reconcile of the sister repo. - -If Sync was clicked and the workload broke: - -- Click **Rollback** in Argo CD UI to the previous synced revision (if there is one). -- Or revert both PRs and re-Sync — the previous state had no `Application`, so the workload is removed. - ---- - -## Related - -- Schema: [custom-values-schema.md](../schemas/custom-values-schema.md), [raw-manifest-sidecar-schema.md](../schemas/raw-manifest-sidecar-schema.md). -- Procedure: [update-chart-version.md](update-chart-version.md) for bumping after onboarding. -- Procedure: [deboard-app.md](deboard-app.md) for retirement. -- Skill: [skills/infra/onboard-app.md](../../../skills/infra/onboard-app.md) — agent-callable wrapper. -- ADR: [ADR-A3-per-cluster-scheduling.md](../../../wiki/analyses/ADR-A3-per-cluster-scheduling.md). diff --git a/docs/platform/procedures/onboard-new-cluster.md b/docs/platform/procedures/onboard-new-cluster.md deleted file mode 100644 index 78caa3f..0000000 --- a/docs/platform/procedures/onboard-new-cluster.md +++ /dev/null @@ -1,140 +0,0 @@ -# Procedure — Onboard a new cluster's overrides - -> **Layer:** Layer 1 — HIGH RISK. -> **Blast radius:** establishing a new deployment target. Mistakes here are repeated for every app subsequently onboarded. -> **Approval:** platform team + cluster owner. CMR mandatory. - -When a new GKE cluster is provisioned (typically by `terraform-gcp-infra` or its successor), this repo gains a new `helm-overrides//` directory and the sister repo gains the `ApplicationSet` (or per-cluster Applications) that route to it. This procedure covers the slice owned by `devops-infra-helm-charts`. - ---- - -## Pre-conditions - -- [ ] The cluster exists in GCP — confirmed by the platform team. -- [ ] The cluster is registered in the relevant Argo CD instance(s). -- [ ] The cluster's `nodeSelector` / `tolerations` / `computeClass` topology is documented (node pool names + taints). -- [ ] The cluster has a `SecretStore` / `ClusterSecretStore` for `external-secrets` — or onboarding `external-secrets` is part of this PR. -- [ ] CMR open. - ---- - -## Steps - -### 1. Confirm the naming convention - -| Pattern | Use | -|---------|-----| -| `k8s--prd-ase1[c]` | Standard BU prod cluster (GCP zone-a or zone-c) | -| `k8s-shared-int-ase1` | Shared int | -| `k8s-aurva-prd-ase1` | Aurva | -| `k8s-supply-dev-ase1` | Dev/sandbox | -| `db--...` | Auto-named dataplane / data-tier | - -The directory name is a **contract** — it must match the cluster name as registered in Argo CD. Deviating by a hyphen or case is a silent bind failure. - -### 2. Determine the cluster's scheduling profile - -| Cluster type | Scheduling key | -|--------------|----------------| -| GKE Autopilot | `cloud.google.com/compute-class` | -| Standard GKE | `dedicated:` | - -Get the actual node-pool / compute-class names from the cluster owner. If GKE Autopilot, also collect the list of `ComputeClass` resources that need to land in `helm-overrides///computeclass/`. - -### 3. Decide the minimal app set - -Most clusters need at least: - -- `external-secrets` — to materialise secrets from GCP Secret Manager -- `kube-state-metrics` — for fleet observability -- `victoria-metrics-agent` — to ship metrics to the central VM -- `fluentd` (or equivalent log shipper) - -Plus per-cluster role: - -- BU clusters: `contour-internal-0`, `contour-internal-1`, sometimes `contour-external` — see [contour-nodeselector-tolerations-summary.md](../../../contour-nodeselector-tolerations-summary.md) for the convention -- DSGPU / Spark clusters: GPU operators, `keda` for queue autoscaling -- Dataplane (`db-*`) clusters: minimal — `kube-state-metrics` + `victoria-metrics-agent` only - -### 4. Create the directory and minimal apps - -```bash -mkdir -p helm-overrides/ -cd helm-overrides/ -``` - -For each app in the minimal set, follow [onboard-app-to-cluster.md](onboard-app-to-cluster.md) (one app per sub-PR after this base PR lands, OR all in one base PR — see step 7). - -### 5. Update the per-cluster Contour matrix - -If the cluster runs Contour, add a section to [contour-nodeselector-tolerations-summary.md](../../../contour-nodeselector-tolerations-summary.md) with the per-instance scheduling. Without this, future Contour edits on the cluster have no reference. - -### 6. (If GKE Autopilot) Add the `ComputeClass` resources - -Each Autopilot Contour / pool needs a corresponding `ComputeClass`. Drop them in: - -``` -helm-overrides///computeclass/-cc.yaml -``` - -Schema: [raw-manifest-sidecar-schema.md §ComputeClass](../schemas/raw-manifest-sidecar-schema.md). The `metadata.name` of each `ComputeClass` must equal the value the corresponding app's `nodeSelector: {cloud.google.com/compute-class: }` references. - -### 7. Open the base PR - -```bash -git checkout -b cluster/-base -git add helm-overrides// -git add contour-nodeselector-tolerations-summary.md # if updated -git commit -git push origin cluster/-base -gh pr create --base main --title "Onboard — base override set" -``` - -PR description must include: - -- Cluster owner approver tag. -- Platform team approver tag. -- CMR ticket. -- Sister-repo PR for the `ApplicationSet` / per-cluster `Application` set. -- The minimal app set being shipped, with per-app rationale. -- The cluster's scheduling profile (key style + sample `nodeSelector`). - -### 8. Pair the sister-repo PR - -In `github.com/Meesho/devops-infra-argo-config`, draft the `ApplicationSet` (or per-cluster Application set) that walks `helm-overrides//`. The sister-repo PR depends on this PR being merged first (otherwise the paths it references don't exist). - -### 9. Stage subsequent app additions - -After the base PR merges: - -- Each additional app (`contour-internal-0`, `flagger`, etc.) is its own PR via [onboard-app-to-cluster.md](onboard-app-to-cluster.md). -- Don't try to land the whole cluster in one PR — review surface explodes; rollback is all-or-nothing. - ---- - -## Anti-patterns - -1. **Naming the directory `` slightly differently** from how Argo CD registered it. Silent bind failure. -2. **Cloning another cluster's directory verbatim.** Per-cluster scheduling differs; `external-secrets` `SecretStore` references differ; image-tag pinning differs. -3. **Onboarding 30 apps in the base PR.** Stage them — base + per-app. -4. **Forgetting the Contour matrix update.** Future Contour edits then have no reference and silently mis-schedule. -5. **Skipping the `ComputeClass` resources** on a GKE Autopilot cluster. `nodeSelector` references a class that doesn't exist; pods Pending. - ---- - -## Rollback - -If the cluster onboarding turns out to be premature: - -- Revert the values-side PR. -- Revert the sister-repo `ApplicationSet` PR. -- The cluster reverts to "no Argo Applications" — workloads that were synced manually before revert remain (Argo CD doesn't `prune` what's not in its scope after the `Application` is removed); for a clean wipe, manually `kubectl delete ns` the affected namespaces. - ---- - -## Related - -- Procedure: [onboard-app-to-cluster.md](onboard-app-to-cluster.md) — for each app after the base PR. -- Schema: [raw-manifest-sidecar-schema.md](../schemas/raw-manifest-sidecar-schema.md) — for `ComputeClass` and `ExternalSecret` sidecars. -- Reference: [contour-nodeselector-tolerations-summary.md](../../../contour-nodeselector-tolerations-summary.md). -- ADR: [ADR-A3-per-cluster-scheduling.md](../../../wiki/analyses/ADR-A3-per-cluster-scheduling.md). diff --git a/docs/platform/procedures/update-chart-version.md b/docs/platform/procedures/update-chart-version.md deleted file mode 100644 index 813888e..0000000 --- a/docs/platform/procedures/update-chart-version.md +++ /dev/null @@ -1,176 +0,0 @@ -# Procedure — Update a chart's pinned dependency version - -> **Layer:** Layer 1 — HIGH RISK (charts in `helm-templates/` are consumed by every cluster that runs them). -> **Blast radius:** every cluster that has an Argo `Application` referencing this chart will pick up the new version on next sync. -> **Approval:** platform team. CMR mandatory for prod-fleet charts. - -In this repo, charts in `helm-templates//` are typically thin wrappers — `Chart.yaml` declares an upstream chart as a dependency, and `Chart.lock` pins the resolved subchart. "Bumping the chart version" means: - -1. Update `dependencies[].version` in `Chart.yaml`. -2. Run `helm dependency update` to refresh `Chart.lock` (and re-pull the subchart). -3. Test render with representative cluster overrides. - ---- - -## When to use this procedure - -- Upgrading a wrapper chart's pinned subchart version (e.g. `argo-cd 7.7.23` → `7.8.0`). -- Following a security advisory in an upstream chart. - -This is **not** the right procedure for: - -- A blue-green migration to a new major version → use [blue-green-chart-migration.md](blue-green-chart-migration.md). -- Editing chart templates → use [fork-upstream-chart.md](fork-upstream-chart.md). -- Changing values without a chart bump → that's per-cluster `custom-values.yaml` work, not this. - ---- - -## Pre-conditions - -- [ ] The new upstream version exists and has a published changelog. -- [ ] You have read the changelog for breaking template / values changes. -- [ ] The bump is the **explicit headline** of the PR (not a side-effect of another change). -- [ ] CMR open if this is a prod-fleet chart (Argo CD, Contour, VictoriaMetrics, ingress, cert-manager). - ---- - -## Steps - -### 1. Read the changelog - -```bash -# For most upstream charts: -helm repo update -helm search repo / --versions | head -20 -# Read the chart's CHANGELOG.md or release notes on GitHub. -``` - -Look specifically for: - -- **Removed values keys** — would silently no-op overrides. -- **Renamed values keys** — same. -- **CRD changes** — might require manual `kubectl apply` of new CRDs. -- **Breaking template changes** (e.g. label selector immutability on Deployments). -- **Required Kubernetes version bumps**. - -If any of these apply, this is not a simple version bump — escalate to a blue-green migration ([blue-green-chart-migration.md](blue-green-chart-migration.md)). - -### 2. Update `Chart.yaml` - -```bash -$EDITOR helm-templates//Chart.yaml -``` - -Find the `dependencies[]` entry and bump `version:`. Example: - -```diff - dependencies: - - name: argo-cd -- version: 7.7.23 -+ version: 7.8.0 - repository: https://argoproj.github.io/argo-helm -``` - -### 3. Refresh `Chart.lock` - -```bash -helm dependency update helm-templates/ -``` - -This: - -- Verifies the new version resolves. -- Updates `Chart.lock` with the new digest. -- Re-pulls the subchart `.tgz` into `helm-templates//charts/`. - -The lockfile must be committed alongside `Chart.yaml`. A bump without a refreshed lockfile is incomplete. - -### 4. Render against a representative cluster's overrides - -Pick a cluster that runs this chart with a non-trivial override set: - -```bash -sibling=$(find helm-overrides -maxdepth 2 -type d -name '' \ - | grep -v '^helm-overrides/db-' | head -1) -helm template helm-templates/ -f "$sibling/custom-values.yaml" | head -120 -``` - -If the render errors → the chart bump introduced a values incompatibility. **Stop.** Either: - -- Find the new values shape and update affected `custom-values.yaml` files in this PR, OR -- Defer the bump until those updates are scoped. - -### 5. Spot-check `helm diff` against a live cluster (recommended) - -```bash -helm diff upgrade helm-templates/ \ - -f helm-overrides///custom-values.yaml \ - --kube-context= -``` - -What you want to see: - -- Image tag bumps. -- Label updates (often `helm.sh/chart`). -- Possibly new CRDs or RBAC. - -What's a red flag: - -- `Deployment` selector changes (immutable; will fail to apply). -- `StatefulSet` `volumeClaimTemplates` changes. -- Resource removals you didn't expect. - -### 6. Commit and open the PR - -```bash -git checkout -b chart-bump/- -git add helm-templates//Chart.yaml helm-templates//Chart.lock -git add helm-templates//charts/ # if the .tgz was re-pulled -git commit -git push origin chart-bump/- -gh pr create --base main --title "chart bump: " -``` - -PR description: - -- Procedure followed: this file. -- Old → new with a link to the upstream changelog. -- The list of clusters that consume this chart (`grep -rl '' helm-overrides | head`). -- The render(s) and diff(s) you ran. -- Whether any breaking-change fallout is bundled (preferably not — if so, scope to a separate PR). - -### 7. After merge — Argo CD picks up the new chart on next sync - -For each cluster running this chart: - -- The cluster's Argo CD detects the chart hash change. -- The `Application` becomes `OutOfSync`. -- A human clicks **Sync** per [ADR-A5](../../../wiki/analyses/ADR-A5-manual-sync-default-for-infra.md). -- Watch rollout per cluster — don't sync 20 clusters simultaneously. - ---- - -## Anti-patterns - -1. **Bumping `Chart.yaml` without refreshing `Chart.lock`.** Argo CD uses the lockfile; the bump silently no-ops. -2. **Bundling a chart bump with values changes.** Two separate PRs — one for the bump, one for any values that need to change because of the bump. -3. **Bumping a major version (e.g. 7.x → 8.x) in place.** Use [blue-green-chart-migration.md](blue-green-chart-migration.md) — give yourself a sibling chart and migrate cluster-by-cluster. -4. **Syncing every cluster simultaneously after merge.** Stagger; one cluster at a time, with a soak between. -5. **Skipping the changelog read.** Surprises you'll regret. - ---- - -## Rollback - -Open a follow-up PR that reverts `Chart.yaml` and `Chart.lock` to the previous versions. After merge, each cluster's Argo CD shows `OutOfSync` against the old chart; click **Sync** to roll back per cluster. - -If the chart bump landed CRDs that are now incompatible with the old version, the rollback may require manual CRD cleanup — coordinate with the platform team. - ---- - -## Related - -- Procedure: [blue-green-chart-migration.md](blue-green-chart-migration.md) — for major-version bumps. -- Procedure: [fork-upstream-chart.md](fork-upstream-chart.md) — when the bump requires template edits. -- Skill: [skills/infra/bump-chart-version.md](../../../skills/infra/bump-chart-version.md). -- ADR: [ADR-A1-cache-vs-upstream-charts.md](../../../wiki/analyses/ADR-A1-cache-vs-upstream-charts.md). diff --git a/docs/platform/runbooks/argocd-sync-failure.md b/docs/platform/runbooks/argocd-sync-failure.md deleted file mode 100644 index 9d49c29..0000000 --- a/docs/platform/runbooks/argocd-sync-failure.md +++ /dev/null @@ -1,231 +0,0 @@ -# Runbook — Argo CD Sync Failure (infra release) - -> **Type:** Decision tree. -> **Entry symptom:** an Argo CD `Application` for an infra release is `OutOfSync`, errored, or stuck `Progressing`. -> **Layer:** mostly Layer 1 (read state, propose YAML diff). Some branches are Layer 2 (advisory). - -This runbook handles sync failures for infra Applications routed by `github.com/Meesho/devops-infra-argo-config` and rendering against this repo's `helm-overrides///`. - ---- - -## Entry — gather context - -```bash -APP= # e.g. argocd, contour-internal-0 -CLUSTER= # e.g. k8s-supply-prd-ase1 -NS=$(argocd app get $APP -o json | jq -r '.spec.destination.namespace') -PROJECT=$(argocd app get $APP -o json | jq -r '.spec.project') - -argocd app get $APP # headline -argocd app get $APP -o json | jq -r '.status.conditions[]?' -argocd app get $APP -o json | jq -r '.status.operationState.message // empty' -``` - -Note which Argo CD instance you're hitting — most infra Applications live in a per-cluster Argo CD install. - ---- - -## Decision tree - -```text -START - │ - └── Is `argocd app get $APP` known to this Argo at all? - │ - ├── NO → §1 — Application not found - │ - └── YES → What's the symptom? - │ - ├── Sync failed with an error message → §2 — Errored sync - ├── Sync stuck `Progressing` for >5 min → §3 — Stuck progressing - ├── App is `OutOfSync` but Sync hasn't run → §4 — OutOfSync only - └── `Synced`+`Healthy` but workload bad → §5 — Wrong workload (leave runbook) -``` - ---- - -## §1 — Application not found - -| Sub-check | Action | -|-----------|--------| -| Are you on the right Argo CD instance? | Most v2 infra apps live in per-cluster Argo CDs. | -| Was the app deboarded recently? | `git -C log --diff-filter=D -- 'apps//*'` and `git log --diff-filter=D -- 'helm-overrides///'`. | -| Did the values directory land on `main`? | `git log --all -- 'helm-overrides///'`. | - -If the file *should* exist on `main` but the `Application` resource isn't created → **Layer 2** — escalate to the platform team. The cluster's Argo CD bootstrap (`ApplicationSet`) may not be picking up the path. - ---- - -## §2 — Errored sync (read the error message) - -### §2a — `repository not accessible / authentication required` - -| Action | -|--------| -| Check `spec.source.repoURL` is `github.com/Meesho/devops-infra-helm-charts.git`. | -| If yes, the credentials in Argo CD's repo-list need refreshing. **Layer 2** — recommend platform team rotates credentials. | - -### §2b — `path 'X' does not exist in repo Y` - -```text -path 'helm-overrides/k8s-supply-prd-ase1/argocd' does not exist -``` - -| Action | -|--------| -| `ls helm-overrides///` on `main`. | -| If absent → either the values-side PR wasn't merged, or the path was typo'd in the sister-repo `Application`. **Layer 1** — open a fix PR (sister repo). | -| If a blue-green migration just landed: the `Application` may be pointing at the **old** chart path that was retired. **Layer 1** — repoint the `Application` to the new sibling path. | - -### §2c — `Helm template error` / `values file not found` - -```text -open helm-overrides///custom-values.yaml: no such file or directory -``` - -| Action | -|--------| -| Verify the file exists on `main`: `git ls-tree origin/main -- helm-overrides///custom-values.yaml`. | -| If absent → onboarding is incomplete. **Layer 1** — open the missing values-side PR. | - -### §2d — `unable to render manifests` / `template error` - -Helm template error inside the chart (missing required value, type mismatch). - -| Action | -|--------| -| Reproduce locally: `helm template helm-templates/ -f helm-overrides///custom-values.yaml`. | -| Determine whether the fix is in this repo (rare — usually values shape changed) or `helm-templates/` (more common after a chart bump). **Layer 1**. | -| If the error is `Cannot use existing release: ...` — see §2h. | - -### §2e — `cluster not found / dial tcp ... no route to host` - -| Action | -|--------| -| **Layer 2** — escalate to platform team. Cluster API server unreachable, or cluster-secret stale in Argo CD. | -| Do **not** edit `spec.destination.{server,name}` to redirect; that masks the underlying cluster issue. | - -### §2f — `forbidden: ...` / admission webhook deny - -```text -admission webhook "validate.kyverno.svc-fail" denied the request -``` - -| Action | -|--------| -| Look at the rule that denied (Kyverno? PSP? OPA? GKE Autopilot policy?). | -| Often the chart's manifest violates a cluster policy (e.g. `runAsUser: 0`, missing `securityContext`). | -| **Layer 1** — fix in chart values; pair with the policy team if the policy is wrong. | -| GKE Autopilot specifically denies many privileged settings — read the deny message carefully. | - -### §2g — `webhook errored: ... cert-manager / external-secrets / kyverno` - -A webhook that should validate the new resource is itself unhealthy. - -| Action | -|--------| -| `kubectl get pods -n cert-manager` (or the relevant operator's namespace) — is it running? | -| **Layer 2** — recommend recovering the webhook before re-syncing this app. | - -### §2h — `cannot patch ... immutable field` - -Most often: `Deployment.spec.selector` or `StatefulSet.volumeClaimTemplates`. A chart bump that changes labels. - -| Action | -|--------| -| Read the upstream changelog to confirm. | -| **Layer 2** — recommend deleting the old `Deployment` / `StatefulSet` (with the workload owner) so the chart can recreate it. **Do not delete blindly** — for `StatefulSet`, the PVCs survive but the rollout is disruptive. | -| For systemic immutable-field changes across a chart bump, this is a sign the bump should have been a [blue-green migration](../procedures/blue-green-chart-migration.md). Roll back, plan the migration. | - -### §2i — `dependent CRD ... not installed` - -The chart needs a CRD that doesn't exist yet on the cluster. - -| Action | -|--------| -| Check whether the chart includes the CRD in `templates/crds/` (most upstream charts ship CRDs). | -| If yes: the chart's `helm template` may not include CRDs by default — Argo CD has `IncludeCRDs` semantics; check the `Application`'s `helm.skipCrds` setting. | -| If the CRD is supposed to come from a different chart (`cert-manager`, `kube-prometheus-stack`): **Layer 2** — sync that chart first. | - ---- - -## §3 — Stuck `Progressing` for > 5 minutes - -The sync started but resources aren't reconciling. - -| Sub-check | Action | -|-----------|--------| -| `argocd app get $APP --refresh` shows resource-level status. | Look for `Progressing` resources. | -| Is a Deployment failing to roll out? | `kubectl rollout status deploy/ -n $NS`. If yes → see [pod-pending-scheduling.md](pod-pending-scheduling.md) or [ingress-down.md](ingress-down.md). | -| Is a Job hung? | `kubectl describe job/ -n $NS`. Old `Job`s sometimes block syncs (Helm pre-/post-install hooks). | -| Is a `PreSync`/`PostSync` hook hanging? | `kubectl get pods -n $NS -l argocd.argoproj.io/hook=PostSync`. | - -If the workload itself is the problem, leave this runbook. - ---- - -## §4 — `OutOfSync` only (no error, sync hasn't run) - -Argo CD sees a diff between git and the cluster. **Most infra apps are intentionally manual-sync** ([ADR-A5](../../../wiki/analyses/ADR-A5-manual-sync-default-for-infra.md)). - -| Sub-check | Action | -|-----------|--------| -| Is this expected? (e.g. you just merged a PR.) | Click Sync. | -| Diff suspicious? (e.g. someone `kubectl edit`-ed.) | `argocd app diff $APP`. If out-of-band edit happened, the GitOps contract was violated; recommend reverting the manual change or capturing it in a PR. | -| Diff has sat for > 1 day? | Notify the app owner — manual-sync apps rot if no one clicks. | - ---- - -## §5 — `Synced` and `Healthy` but workload misbehaving - -Argo thinks all is fine; the workload is broken. Not a sync failure. **Leave this runbook.** - -| Symptom | Where to go | -|---------|-------------| -| Pods crashlooping | [pod-pending-scheduling.md](pod-pending-scheduling.md) §3 | -| Ingress 5xx | [ingress-down.md](ingress-down.md) | -| Specific feature broken | App-team playbook | - ---- - -## §6 — Special: blue-green migration in flight - -If this app is in a `` ↔ `-` migration: - -- Confirm which variant the `Application` points at (check `spec.source.path`). -- The chart name may have changed in the new variant; release-name pinning via `fullnameOverride` may be required to keep the same Service DNS during cutover. -- A failed sync mid-migration is the trigger to roll back (`spec.source.path` ← old) and Sync, not to push forward. -- Read [blue-green-chart-migration.md](../procedures/blue-green-chart-migration.md) before deciding. - ---- - -## Escalation matrix - -| Symptom | Action | Escalate to | -|---------|--------|-------------| -| §1 + bootstrap looks healthy | Investigate further | App owner | -| §2a (repo auth) | Confirm allowed repoURL; rotate creds | Platform team | -| §2b (path missing) | Fix in this repo or sister repo | App owner | -| §2c, §2d (helm render) | Reproduce; fix values or chart | App owner / platform team | -| §2e (cluster unreachable) | Don't edit destination | Platform team | -| §2f (admission webhook) | Fix in chart values | App + policy team | -| §2g (webhook unhealthy) | Recover the webhook first | Platform team | -| §2h (immutable field) | Probably needs blue-green | Platform team | -| §3 (stuck > 30 min) | Check pod events; consider workload rollback | App owner | - ---- - -## Done conditions - -- `argocd app get $APP` shows `Synced` + `Healthy`. -- The PR or manual fix that resolved it is on `main`. -- If the failure was caused by a regression, a postmortem / RCA is scheduled. - ---- - -## Related - -- Runbook: [pod-pending-scheduling.md](pod-pending-scheduling.md). -- Runbook: [ingress-down.md](ingress-down.md). -- Schema: [custom-values-schema.md](../schemas/custom-values-schema.md). -- Boundaries: [AGENT_BOUNDARIES.md](../../global/AGENT_BOUNDARIES.md). diff --git a/docs/platform/runbooks/ingress-down.md b/docs/platform/runbooks/ingress-down.md deleted file mode 100644 index 63e30a2..0000000 --- a/docs/platform/runbooks/ingress-down.md +++ /dev/null @@ -1,197 +0,0 @@ -# Runbook — Ingress (Contour) is down on a cluster - -> **Type:** Decision tree. -> **Entry symptom:** services on a cluster are returning 5xx, not reachable, or DNS doesn't resolve to working endpoints. -> **Layer:** mostly Layer 2 (advisory — recommend kubectl actions). Layer 1 only when the fix is a values diff in this repo. - -Contour is the ingress for almost every BU cluster, and most clusters run **multiple Contour releases** — `contour-external`, `contour-external-1`, `contour-internal-0`, `contour-internal-1`, `contour-internal-intra-{0,1}`. Each maps to a different node pool / dedicated taint or compute class. The matrix is in [contour-nodeselector-tolerations-summary.md](../../../contour-nodeselector-tolerations-summary.md). **Read it before editing any Contour values.** - ---- - -## Entry — gather context - -```bash -CLUSTER= -CTX= - -# Which Contour instances does this cluster run? -ls helm-overrides/$CLUSTER | grep '^contour' - -# Pod state for each Contour -for c in $(ls helm-overrides/$CLUSTER | grep '^contour'); do - echo "=== $c ===" - kubectl --context=$CTX get pods -n projectcontour -l app.kubernetes.io/instance=$c -o wide 2>/dev/null \ - || kubectl --context=$CTX get pods --all-namespaces -l app.kubernetes.io/instance=$c -o wide -done -``` - ---- - -## Decision tree - -```text -START - │ - └── Which Contour instance is affected? - │ - ├── External (`contour-external*`) → §A — North-south traffic - ├── Internal (`contour-internal-*`) → §B — East-west traffic - └── Both / unsure → §C — Cluster-wide (worst case) -``` - -For each branch: - -```text - └── What's the failure shape? - │ - ├── Pods Pending → §1 — Scheduling failure - ├── Pods CrashLooping → §2 — Contour boot failure - ├── Pods Running but no endpoints → §3 — Service / load-balancer unhealthy - ├── HTTPS responses are 5xx → §4 — Backend / cert / config error - └── DNS doesn't resolve → §5 — external-dns / DNS plumbing -``` - ---- - -## §1 — Contour pods Pending - -This is almost always a scheduling-key mismatch. **The single most common Contour incident.** - -```bash -kubectl --context=$CTX describe pod -``` - -Read the `Events:` section. Common causes: - -| Reason | Fix in | -|--------|--------| -| `0/N nodes available: 1 node(s) had untolerated taint =` | Tolerations in `helm-overrides///custom-values.yaml`. Cross-reference [contour-nodeselector-tolerations-summary.md](../../../contour-nodeselector-tolerations-summary.md). **Layer 1.** | -| `0/N nodes available: 1 node(s) didn't match Pod's node affinity/selector` | `nodeSelector` in the values. Same fix. **Layer 1.** | -| `0/N nodes available: 1 node(s) had no available compute class` (GKE Autopilot) | Either the `ComputeClass` resource is missing, or `nodeSelector: cloud.google.com/compute-class: ` references a class that doesn't exist. Check `helm-overrides///computeclass/*-cc.yaml`. **Layer 1.** | -| `0/N nodes available: insufficient cpu/memory` | Node-pool autoscaler is at max. **Layer 2** — escalate to cluster owner / platform team. | - -The fix is always to bring the values' `nodeSelector` / `tolerations` into agreement with the cluster's actual node-pool topology. Never guess — read the matrix and the cluster's other apps. - ---- - -## §2 — Contour pods CrashLooping - -```bash -kubectl --context=$CTX logs -c contour --previous -kubectl --context=$CTX logs -c envoy --previous -``` - -| Pattern | Cause | Fix in | -|---------|-------|--------| -| `failed to load TLS certificates` | Missing `Secret`, expired cert, wrong key | `cert-manager` / `external-secrets`. **Layer 2.** | -| `error parsing config: invalid HTTPProxy` | A user `HTTPProxy` is malformed and Contour refuses to load | The user's namespace. **Layer 2** — recommend `kubectl get httpproxy -A` to find the offender, then fix in the consuming team's repo. | -| `bind: address already in use` | Two Contour pods on the same node fighting for the host port | Pod anti-affinity in values. **Layer 1.** | -| `error: ratelimit service ... unavailable` | `contour-rate-limit` sidecar / external service is down | Operator action. **Layer 2.** | - ---- - -## §3 — Contour Running but no endpoints / LB unhealthy - -The Contour pods are healthy, but downstream LB / DNS / Service routing is broken. - -```bash -kubectl --context=$CTX get svc -n projectcontour -kubectl --context=$CTX get endpoints -n projectcontour -kubectl --context=$CTX describe svc -n projectcontour -``` - -| Sub-check | Action | -|-----------|--------| -| `Service` of type `LoadBalancer` has no `EXTERNAL-IP` | GCP LB provisioning failed. Check the `Service` annotations match the cluster's LB pattern. **Layer 2** — escalate. | -| `Endpoints` empty | The Contour pods aren't matching the `Service` selector. Often a label drift between values and the chart's defaults. **Layer 1** — fix selectors. | -| `Service` annotations mention `cloud.google.com/load-balancer-type: Internal` but external traffic is expected | Wrong Contour instance / Service annotation. **Layer 1.** | -| `Health checks failing` on the GCP LB | Backend pods aren't ready. See §4. | - ---- - -## §4 — HTTPS responses are 5xx - -Don't curl the production endpoint yourself — that violates [SANCTITY_RULES R3](../../global/SANCTITY_RULES.md). Instead, recommend the user check from a controlled vantage: - -```bash -# From inside the cluster -kubectl --context=$CTX run -it --rm curl-test --image=curlimages/curl --restart=Never -- \ - curl -v https://..svc.cluster.local - -# Contour access logs -kubectl --context=$CTX logs -c envoy | tail -50 -``` - -| Pattern | Cause | Fix in | -|---------|-------|--------| -| `503 no_healthy_upstream` | Backend pods all unhealthy | App's pod readiness — see [pod-pending-scheduling.md](pod-pending-scheduling.md). **Layer 2.** | -| `502 upstream connect error` | Backend connection refused / TLS mismatch | App config / mTLS. **Layer 2.** | -| `404 route not found` | No matching `HTTPProxy`/`Ingress` | The user's `HTTPProxy` is missing or has the wrong host/path. **Layer 2.** | -| `500` from the app | Application error | App-team playbook. **Out of scope for this runbook.** | - ---- - -## §5 — DNS doesn't resolve - -| Sub-check | Action | -|-----------|--------| -| Is `external-dns` running on this cluster? | `kubectl --context=$CTX get pods -n external-dns`. | -| Are there `Service` resources with `external-dns.alpha.kubernetes.io/hostname` annotations? | `kubectl --context=$CTX get svc -A -o json \| jq '.items[] \| select(.metadata.annotations."external-dns.alpha.kubernetes.io/hostname")'`. | -| Did `external-dns` reconcile recently? | `kubectl --context=$CTX logs deploy/external-dns -n external-dns \| tail -30`. | -| Is the Cloud DNS zone wired up? | **Layer 2** — out of scope; escalate to platform team. | - ---- - -## §C — Cluster-wide ingress outage - -If both internal and external Contour are degraded simultaneously: - -1. **Stop. Don't iterate values fixes.** This is incident-grade. -2. **Layer 2 — escalate to platform team immediately.** -3. Check whether a recent merge in this repo or the sister repo correlates: `git log --since='2 hours ago' -- helm-overrides/$CLUSTER/contour*` and the same in `devops-infra-argo-config`. -4. If a recent merge is implicated: revert it, click Sync to roll back to the previous state. -5. If no recent merge: cluster-level issue (node pool, network policy, GCP LB) — out of scope for this repo. - ---- - -## When this repo *is* the right place to fix - -A Contour outage traces back to `devops-infra-helm-charts` only when: - -1. **`nodeSelector` / `tolerations` / `computeClass`** were copied from the wrong cluster. -2. **A chart bump** introduced an immutable-selector change or removed a values key. -3. **A blue-green migration** was cutover prematurely (Application points at `-vX.Y.Z` but values are still on the old shape). -4. **`fullnameOverride`** was changed (very rare, but catastrophic). - -For 80%+ of ingress incidents, the fix is **outside** this repo (cluster issues, app-side issues, GCP LB, DNS). - ---- - -## Escalation matrix - -| Symptom | First responder | Escalate to | -|---------|-----------------|-------------| -| §1 (pending — scheduling) | Yourself with the values fix | Cluster owner if node pool full | -| §2 (CrashLoop — TLS) | cert-manager team | Security if cert source unknown | -| §2 (CrashLoop — invalid HTTPProxy) | Consuming team | Platform if Contour itself is broken | -| §3 (no endpoints / LB) | Cluster owner | Platform team | -| §4 (5xx) | App team | Platform team if Contour-side | -| §5 (DNS) | Platform team | — | -| §C (cluster-wide outage) | Platform team — pager | — | - ---- - -## Done conditions - -- `kubectl get pods -n projectcontour` shows N/N Ready for the affected instances. -- Synthetic probes from the app team return expected status codes. -- If the root cause was in this repo, the fix is on `main` and synced. - ---- - -## Related - -- Reference: [contour-nodeselector-tolerations-summary.md](../../../contour-nodeselector-tolerations-summary.md). -- Runbook: [pod-pending-scheduling.md](pod-pending-scheduling.md). -- Runbook: [argocd-sync-failure.md](argocd-sync-failure.md). -- ADR: [ADR-A3-per-cluster-scheduling.md](../../../wiki/analyses/ADR-A3-per-cluster-scheduling.md). diff --git a/docs/platform/runbooks/metrics-gap.md b/docs/platform/runbooks/metrics-gap.md deleted file mode 100644 index 4ec7ab9..0000000 --- a/docs/platform/runbooks/metrics-gap.md +++ /dev/null @@ -1,233 +0,0 @@ -> Per AI Blitz Plan §platform.runbooks. Layer: 1. Repo: devops-infra-helm-charts. - -# Runbook — Metrics gap on a cluster / namespace - -> **Type:** Decision tree. -> **Entry symptom:** "Grafana panels are blank for `` / `` / ``," or an alert that should be firing isn't, or VictoriaMetrics shows `no data`. -> **Layer:** mostly Layer 2 (advisory). Layer 1 only when the fix is a values diff in this repo. - -The observability path on Meesho's GKE fleet is: - -``` -workload Pod (exposes /metrics) - └→ victoria-metrics-agent (vmagent) scrapes - └→ remote_write to victoria-metrics-cluster (vmstorage) - └→ vmselect ← Grafana / vmalert query -``` - -A metrics gap can be at any hop. Walk this tree top-down — the most common root cause is hop 1 (scrape config). - ---- - -## Entry — gather context - -```bash -CLUSTER= -CTX= -NS= -METRIC= - -# Confirm the cluster runs VM agent + cluster -ls helm-overrides/$CLUSTER | grep -E '^victoria-metrics-(agent|cluster)' - -# Confirm pods are healthy -kubectl --context=$CTX -n monitoring get pods -l app.kubernetes.io/name=victoria-metrics-agent -o wide -kubectl --context=$CTX -n monitoring get pods -l app.kubernetes.io/name=vmstorage -o wide -``` - -If pods are missing/CrashLooping → that's the gap. Skip to §5. - ---- - -## Decision tree - -```text -START - │ - ├── Is the metric known to be emitted by the workload? - │ ├── No → §0 — Workload not emitting; out of scope (app team) - │ └── Yes → - │ - ├── §1 — Is vmagent scraping the workload? - │ ├── No → fix scrape config (Layer 1) - │ └── Yes → - │ - ├── §2 — Are scrape targets healthy (status=up)? - │ ├── Down → fix endpoint reachability (Layer 2 / Layer 1) - │ └── Up → - │ - ├── §3 — Are relabel rules dropping the metric? - │ ├── Yes → adjust relabel_configs (Layer 1) - │ └── No → - │ - ├── §4 — Is remote_write succeeding? - │ ├── No → vmagent → vmstorage path broken (Layer 2 / Layer 1) - │ └── Yes → - │ - ├── §5 — Is vmstorage healthy and ingesting? - │ ├── No → vmstorage outage (Layer 2) - │ └── Yes → - │ - └── §6 — Is the Grafana datasource / tenant correct? - └── Misrouted query → fix datasource URL or tenant header (Layer 1) -``` - ---- - -## §0 — Workload not emitting - -Out of scope for this repo. Confirm with: - -```bash -# Port-forward and curl /metrics directly -kubectl --context=$CTX -n $NS port-forward 9090: & -curl -s localhost:9090/metrics | grep -i "$METRIC" -``` - -If `/metrics` is empty or doesn't contain `$METRIC` → app team. Stop. - ---- - -## §1 — Is vmagent scraping the workload? - -```bash -# vmagent UI exposes /api/v1/targets -kubectl --context=$CTX -n monitoring port-forward svc/victoria-metrics-agent 8429:8429 & -curl -s localhost:8429/api/v1/targets | jq '.data.activeTargets[] | select(.labels.namespace=="'$NS'")' -``` - -If no targets for `$NS`: - -- Look at `helm-overrides/$CLUSTER/victoria-metrics-agent/custom-values.yaml`. -- Check the `additionalScrapeConfigs` (or `config.scrape_configs`) for a job that matches the workload's labels / namespace selector. -- Common cause: a `kubernetes_sd_configs` `namespaces.names` filter excludes `$NS`. -- Common cause: a missing `Pod`/`Service`/`PodMonitor` annotation `prometheus.io/scrape: "true"` on the workload. - -**Layer 1 fix** (if the gap is in the scrape config): edit `helm-overrides/$CLUSTER/victoria-metrics-agent/custom-values.yaml` per [../procedures/modify-observability-config.md](../procedures/modify-observability-config.md). - -**Layer 2 fix** (if the gap is on the workload — missing annotation): hand off to app team. - ---- - -## §2 — Targets healthy? - -```bash -# In the same vmagent /targets output -curl -s localhost:8429/api/v1/targets | jq '.data.activeTargets[] | select(.health!="up") | {labels, lastError}' -``` - -If targets are `down`: - -| `lastError` | Cause | Fix | -|-------------|-------|-----| -| `connection refused` | Workload not listening on declared port | App team — Layer 2. | -| `i/o timeout` | NetworkPolicy / firewall blocking vmagent → workload | Check `NetworkPolicy` in `$NS`. Often a NetworkPolicy allowing only intra-namespace traffic and not vmagent's namespace. **Layer 2** — recommend the workload team allow vmagent. | -| `x509: certificate signed by unknown authority` | mTLS misconfigured | App team — Layer 2. | -| `404 Not Found` | Wrong path (default `/metrics` vs custom) | Add `metrics_path:` in scrape config. **Layer 1.** | - ---- - -## §3 — Relabel rules dropping the metric? - -```bash -yq e '.config.scrape_configs[].metric_relabel_configs, .config.scrape_configs[].relabel_configs' \ - helm-overrides/$CLUSTER/victoria-metrics-agent/custom-values.yaml -``` - -Look for: - -- `action: drop` rules with regexes matching `$METRIC`. -- `action: keep` rules whose regex *excludes* `$METRIC`. -- `action: labeldrop` removing a label the query uses. - -Test in vmagent's UI under the *Targets* tab — it shows the labels post-relabel. - -**Layer 1 fix:** loosen the relabel rule. Re-PR per [../procedures/modify-observability-config.md](../procedures/modify-observability-config.md). - ---- - -## §4 — Remote_write succeeding? - -```bash -kubectl --context=$CTX -n monitoring logs deploy/victoria-metrics-agent | grep -i 'remote_write\|error\|failed' | tail -20 -``` - -| Symptom | Cause | Fix | -|---------|-------|-----| -| `429 Too Many Requests` from vmstorage | vmstorage ingest saturated | Scale vmstorage / vminsert (Layer 1 — see [observability.md](../../global/coding-guidelines/observability.md)). | -| `connection refused` to vmstorage URL | vmstorage Service down or wrong URL | Verify `remoteWrite.url` in vmagent values matches the live `vmstorage` Service DNS. Layer 1. | -| `out of bounds timestamp` | Clock skew on vmagent's node | Layer 2 — node time sync. | -| `series limit exceeded` | Cardinality bomb on vmstorage tenant | Layer 1 — drop the offending label. See [observability.md §Cardinality](../../global/coding-guidelines/observability.md). | - ---- - -## §5 — vmstorage healthy? - -```bash -kubectl --context=$CTX -n monitoring get pods -l app.kubernetes.io/name=vmstorage -o wide -kubectl --context=$CTX -n monitoring describe pod vmstorage-0 | tail -30 -kubectl --context=$CTX -n monitoring exec vmstorage-0 -- df -h /storage -``` - -Failure modes: - -| Symptom | Layer | Fix | -|---------|-------|-----| -| Pod CrashLooping with `out of disk` | Layer 1 | Bump `persistence.size`. Note: PVC growth requires the StorageClass to support `allowVolumeExpansion: true`. See [../schemas/storageclass-priorityclass-schema.md](../schemas/storageclass-priorityclass-schema.md). | -| Pod CrashLooping with retention/index errors | Layer 2 | Escalate — may need data-side intervention. | -| Pod Pending | Layer 1 | Scheduling issue. See [pod-pending-scheduling.md](pod-pending-scheduling.md) and [../../../wiki/analyses/ADR-A3-per-cluster-scheduling.md](../../../wiki/analyses/ADR-A3-per-cluster-scheduling.md). | -| Pod Running but vmselect can't reach it | Layer 1 / 2 | Verify the headless Service and StatefulSet pod-DNS records. | - ---- - -## §6 — Grafana datasource / tenant correct? - -```bash -yq e '.datasources.datasources.yaml.datasources[] | select(.name == "*VictoriaMetrics*" or .type == "prometheus")' \ - helm-overrides/$CLUSTER/grafana/custom-values.yaml -``` - -| Sub-check | Action | -|-----------|--------| -| `url:` points at the correct in-cluster vmselect Service DNS | If wrong, **Layer 1** — fix the values per [../procedures/modify-observability-config.md](../procedures/modify-observability-config.md). | -| `httpHeaderName1: X-Scope-OrgID` (multi-tenant clusters only) | If the cluster runs multi-tenant VM, the tenant header must be set. Layer 1 fix. | -| Datasource `url:` points at an external `*.meeshogcp.in` host | Forbidden — see [../../global/SANCTITY_RULES.md](../../global/SANCTITY_RULES.md) R3. Repoint at in-cluster Service DNS. | - ---- - -## Remediation summary - -| Hop | Likely fix | Layer | Procedure | -|-----|-----------|-------|-----------| -| §1 scrape | Add scrape config / fix selector | 1 | [../procedures/modify-observability-config.md](../procedures/modify-observability-config.md) | -| §2 endpoint | Workload-side / NetworkPolicy | 2 | Hand off to app team | -| §3 relabel | Loosen drop rule | 1 | [../procedures/modify-observability-config.md](../procedures/modify-observability-config.md) | -| §4 remote_write | Scale vmstorage / fix URL | 1/2 | [../procedures/modify-observability-config.md](../procedures/modify-observability-config.md) | -| §5 vmstorage | PVC grow / scheduling fix | 1/2 | [pod-pending-scheduling.md](pod-pending-scheduling.md) | -| §6 datasource | Fix Grafana datasource | 1 | [../procedures/modify-observability-config.md](../procedures/modify-observability-config.md) | - ---- - -## Escalation triggers - -- §5 vmstorage outage with no obvious values-side fix → platform team (observability owner). -- Multi-cluster simultaneous metrics gap → platform team — cluster-level / control-plane issue. -- Cardinality explosion impacting vmstorage stability → platform team + workload owner — joint fix. - ---- - -## Done conditions - -- The query that was returning `no data` returns the expected series in Grafana Explore. -- No remote_write errors in vmagent logs for at least 5 minutes post-fix. -- Alerts that depend on the metric have transitioned from `pending` / silent back to expected state. - ---- - -## Related - -- Procedure: [../procedures/modify-observability-config.md](../procedures/modify-observability-config.md). -- Procedure: [../procedures/modify-alert-rules.md](../procedures/modify-alert-rules.md) — if the gap is "alert silent" not "metric missing." -- Coding guideline: [../../global/coding-guidelines/observability.md](../../global/coding-guidelines/observability.md). -- Runbook: [pod-pending-scheduling.md](pod-pending-scheduling.md) — for §5 vmstorage scheduling failures. -- Runbook: [argocd-sync-failure.md](argocd-sync-failure.md) — if a sync didn't take. diff --git a/docs/platform/runbooks/pod-pending-scheduling.md b/docs/platform/runbooks/pod-pending-scheduling.md deleted file mode 100644 index 2953bb4..0000000 --- a/docs/platform/runbooks/pod-pending-scheduling.md +++ /dev/null @@ -1,192 +0,0 @@ -# Runbook — Pod Pending / wrong-node scheduling - -> **Type:** Decision tree. -> **Entry symptom:** an infra workload's pods are `Pending` indefinitely, or scheduling onto the wrong node pool. -> **Layer:** mostly Layer 2 (advisory — recommend kubectl). Layer 1 when the fix is a values edit here. - -The single most common values-side bug in this repo is **`nodeSelector` / `tolerations` / `computeClass` copied from the wrong cluster**. ([SANCTITY_RULES R5](../../global/SANCTITY_RULES.md)) - ---- - -## Entry — gather context - -```bash -APP= -CLUSTER= -CTX= -NS= - -kubectl --context=$CTX get pods -n $NS -o wide -kubectl --context=$CTX describe pod -n $NS | tail -40 # Events: section -``` - ---- - -## Decision tree - -```text -START - │ - └── What state are the pods in? - │ - ├── Pending — never scheduled → §1 — Pending pods - ├── ContainerCreating long → §1 — Pending pods - ├── Running but on the WRONG node pool → §2 — Wrong-pool scheduling - ├── ImagePullBackOff / ErrImagePull → §3 — Image pull - ├── CrashLoopBackOff → §4 — Crash loop - ├── Running but PVC unbound → §5 — Storage - └── Running fine → leave runbook -``` - ---- - -## §1 — Pending pods - -```bash -kubectl --context=$CTX describe pod -n $NS -``` - -Read the `Events:` section. Common patterns: - -| Reason | Diagnosis | Fix in | -|--------|-----------|--------| -| `0/N nodes available: 1 node(s) had untolerated taint {key: dedicated, value: , effect: NoSchedule}` | Pod has wrong toleration or no toleration. | `helm-overrides///custom-values.yaml` `tolerations:`. **Layer 1.** | -| `0/N nodes available: 1 node(s) didn't match Pod's node affinity/selector` | Pod's `nodeSelector` doesn't match any node label. | Values `nodeSelector:`. **Layer 1.** | -| `0/N nodes available: ... had no available compute class` (Autopilot) | `cloud.google.com/compute-class: ` references a `ComputeClass` that doesn't exist on the cluster. | (a) Add the `ComputeClass` resource under `helm-overrides///computeclass/`, or (b) use the right class name. **Layer 1.** | -| `0/N nodes available: insufficient cpu` / `insufficient memory` | Node-pool autoscaler at max, or pod's `requests` too high. | Either reduce `resources.requests`, or **Layer 2** — escalate to cluster owner to raise node-pool max. | -| `0/N nodes available: pod has unbound immediate PersistentVolumeClaims` | PVC is `Pending`. | See §5. | -| `0/N nodes available: didn't tolerate node-pressure taint` | Node has `node.kubernetes.io/disk-pressure` etc. | **Layer 2** — cluster-level issue. | -| `volume "X" not found` | PVC bound to a non-existent PV. | See §5. | - -**The diagnostic for "wrong cluster's scheduling values":** - -```bash -# What does the pod's nodeSelector say? -kubectl --context=$CTX get pod -n $NS -o yaml \ - | yq e '.spec.nodeSelector' - -# What labels do the cluster's nodes actually have? -kubectl --context=$CTX get nodes --show-labels | head -3 - -# Cross-reference: is the cluster's key style 'dedicated:' or 'cloud.google.com/compute-class'? -grep -h 'dedicated:\|cloud.google.com/compute-class' \ - helm-overrides/$CLUSTER/*/custom-values.yaml | sort -u | head -``` - -If the values use `dedicated:` but the cluster only has `cloud.google.com/compute-class:` keys (or vice versa), the values were copied from a sibling cluster. **Author the values from scratch** using the cluster's own key style. - ---- - -## §2 — Running but on the wrong node pool - -The pod scheduled, but on a node it shouldn't be on (e.g. a Contour-internal pod landed on the Contour-external pool). - -| Sub-check | Action | -|-----------|--------| -| What `nodeSelector` does the pod actually have? | `kubectl --context=$CTX get pod -o yaml \| yq e '.spec.nodeSelector'`. | -| Where is it running? | `kubectl --context=$CTX get pod -o wide` — note the `NODE`. Check that node's labels. | -| Is the values-side `nodeSelector` too permissive? | If the chart's default merges with your override, you may have inherited an unintended key. Render with `helm template` and inspect. | - -The fix is to make the `nodeSelector` selective enough that only the intended pool matches. **Layer 1.** - ---- - -## §3 — Image pull failing - -```text -ErrImagePull / ImagePullBackOff -``` - -| Sub-check | Action | -|-----------|--------| -| Is the image pinned to Meesho's GAR mirror? | `kubectl get deploy -n $NS -o jsonpath='{.spec.template.spec.containers[].image}'`. If it's Docker Hub / Quay / GCR upstream, that's [SANCTITY_RULES R11](../../global/SANCTITY_RULES.md) violation. **Layer 1** — fix the image reference. | -| Does the tag exist in the registry? | Out-of-band check (registry UI). | -| Is the registry-pull credential present on the cluster? | `kubectl get secret -n $NS \| grep gcr-pull`. **Layer 2** if missing. | - -The image tag is set in `image.repository` / `image.tag` of `custom-values.yaml`. **Layer 1.** - ---- - -## §4 — CrashLoopBackOff - -```bash -kubectl --context=$CTX logs -n $NS --previous -kubectl --context=$CTX describe pod -n $NS -``` - -| Pattern | Likely cause | Fix in | -|---------|--------------|--------| -| Application stack trace, missing config | App expects an env var / file that isn't there | `custom-values.yaml` config section. **Layer 1.** | -| `connection refused` to a dependency | Dependency not up; or wrong DNS | Dependency team. **Layer 2.** | -| `exit code 137` | OOMKilled — `kubectl describe` confirms | Bump `resources.limits.memory` in values. **Layer 1.** | -| `permission denied` on a file | Volume mount / `securityContext` | Values. **Layer 1.** | -| `existing Secret not found` | `existingSecret:` references a secret that doesn't exist | Either fix the name, or add the `ExternalSecret` to `helm-overrides//external-secrets/`. **Layer 1.** | -| Init container failed | Init logs explain | `kubectl --context=$CTX logs -c -n $NS`. | - ---- - -## §5 — Running but PVC unbound - -```bash -kubectl --context=$CTX get pvc -n $NS -kubectl --context=$CTX describe pvc -n $NS -``` - -| Sub-check | Action | -|-----------|--------| -| `Events: Failed to provision volume with StorageClass ""` | The StorageClass doesn't exist on this cluster. | `ls manifests/storageclass/.yaml`. If absent, fix `persistence.storageClass:` in values to a real class. **Layer 1.** | -| `Events: ProvisioningFailed: googleapi: Error 403` | CSI driver lacks IAM permission. **Layer 2.** | Platform / IAM team. | -| PVC `Pending` with no events | StorageClass has `volumeBindingMode: WaitForFirstConsumer` and the consuming pod hasn't been scheduled. | Schedule the pod (resolve §1 first). | -| PVC bound but pod can't mount | Often the access mode mismatch (`ReadWriteOnce` PVC referenced by a multi-replica `Deployment`). | Either set replicas to 1, or use a `StatefulSet` chart variant, or use a Filestore-backed `ReadWriteMany` class. **Layer 1.** | - ---- - -## §6 — When this repo *is* the right place to fix - -For these cases, the fix is a values diff in `helm-overrides///custom-values.yaml`: - -1. Wrong-cluster `nodeSelector` / `tolerations` / `computeClass`. -2. Image tag pointing outside the GAR mirror. -3. `resources.requests` / `limits` mis-sized (OOMKill, throttling). -4. `persistence.storageClass` referencing a non-existent class. -5. `existingSecret:` referencing a non-existent secret. -6. `replicaCount` set on an HPA-managed release. - -For these cases, the fix is **outside** this repo: - -- Node pool full → cluster-owner / Terraform. -- Image not in GAR → image-mirror automation / build pipeline. -- CSI provisioning failures → platform / IAM team. -- App-internal crashes (config, dependencies) → app team. - ---- - -## Escalation matrix - -| Symptom | First responder | Escalate to | -|---------|-----------------|-------------| -| §1 — wrong scheduling values | Yourself with values fix | Cluster owner if topology is unclear | -| §1 — node pool full | Cluster owner | Platform team if quota raise needed | -| §3 — image pull (mirror miss) | Yourself with values fix | Platform if mirror push is missing | -| §4 — OOMKilled | Yourself with `resources.limits` bump | App team if root cause is leak | -| §4 — secret missing | Yourself with `ExternalSecret` add | Security if cluster-level `SecretStore` is missing | -| §5 — CSI provision failure | Platform team | — | -| §C — cluster-wide scheduling failure | Platform team — pager | — | - ---- - -## Done conditions - -- `kubectl rollout status deploy/ -n $NS` returns "successfully rolled out". -- `kubectl get pods -n $NS` shows N/N Ready for the expected replica count. -- Pods scheduled onto the **intended** node pool (verify `kubectl get pods -o wide` shows the right node names). -- If root cause was in this repo, the fix is on `main` and synced. - ---- - -## Related - -- Reference: [contour-nodeselector-tolerations-summary.md](../../../contour-nodeselector-tolerations-summary.md). -- Runbook: [argocd-sync-failure.md](argocd-sync-failure.md). -- Runbook: [ingress-down.md](ingress-down.md). -- Schema: [custom-values-schema.md](../schemas/custom-values-schema.md), [storageclass-priorityclass-schema.md](../schemas/storageclass-priorityclass-schema.md). diff --git a/docs/platform/runbooks/vault-unavailable.md b/docs/platform/runbooks/vault-unavailable.md deleted file mode 100644 index 2aab0a9..0000000 --- a/docs/platform/runbooks/vault-unavailable.md +++ /dev/null @@ -1,251 +0,0 @@ -> Per AI Blitz Plan §platform.runbooks. Layer: 1. Repo: devops-infra-helm-charts. - -# Runbook — Vault unavailable / External Secrets failing to render - -> **Type:** Decision tree. -> **Entry symptom:** Pods are CrashLooping referencing a missing `Secret`, or `kubectl describe externalsecret` shows `SecretSyncedError`, or `kubectl get secrets ` returns NotFound for a name an `ExternalSecret` should be creating. -> **Layer:** mostly Layer 2 (advisory). Layer 1 only when the fix is a values diff in this repo. -> -> **Important Layer 3 boundary:** **Vault HA itself is Layer 3.** Vault runs in production-only with no lower environment, so the agent must NOT attempt server-side fixes (unsealing, leader-election toggling, raft config changes, restoring from snapshot). Those are platform-team / security-team operations. The agent's role here is *diagnose, narrow root cause, and escalate*. - ---- - -## Architecture refresher - -``` -workload Pod (mounts Secret ) - ↑ created by -External Secrets Operator (ESO) controller - │ - └─ reads SecretStore / ClusterSecretStore - │ - ├─ kind: gcpsm → GCP Secret Manager - │ └─ auth via Workload Identity (KSA → GSA binding) - └─ kind: vault → Vault HA cluster - └─ auth via Kubernetes auth (KSA token review) -``` - -Most clusters in the fleet use **GCP Secret Manager** as the primary backend (via Workload Identity), with Vault HA as the secondary for legacy services. Some clusters use Vault as the primary. Confirm before troubleshooting: - -```bash -ls helm-overrides//external-secrets/ -yq e '.spec.provider' helm-overrides//external-secrets/*.yaml 2>/dev/null -``` - ---- - -## Entry — gather context - -```bash -CLUSTER= -CTX= -NS= -ES= - -# ESO controller status -kubectl --context=$CTX -n external-secrets get pods -kubectl --context=$CTX -n external-secrets logs deploy/external-secrets | tail -50 - -# The failing ExternalSecret -kubectl --context=$CTX -n $NS describe externalsecret $ES -kubectl --context=$CTX -n $NS get externalsecret $ES -o yaml | yq e '.status' - -# The SecretStore / ClusterSecretStore it references -STORE=$(kubectl --context=$CTX -n $NS get externalsecret $ES -o jsonpath='{.spec.secretStoreRef.name}') -KIND=$(kubectl --context=$CTX -n $NS get externalsecret $ES -o jsonpath='{.spec.secretStoreRef.kind}') -kubectl --context=$CTX get $KIND $STORE -o yaml | yq e '.status, .spec.provider' -``` - ---- - -## Decision tree - -```text -START - │ - ├── §1 — Is ESO controller running and healthy? - │ ├── No → §1a — ESO outage - │ └── Yes → - │ - ├── §2 — Is the SecretStore / ClusterSecretStore Ready? - │ ├── No → §2a — Store config / auth broken - │ └── Yes → - │ - ├── §3 — Does the ExternalSecret reference a real remote key? - │ ├── No → §3a — Bad spec.data[].remoteRef.key (Layer 1 likely) - │ └── Yes → - │ - ├── §4 — Is the auth path working? (WI binding or Vault K8s auth) - │ ├── No → §4a — Identity binding (Layer 2/3) - │ └── Yes → - │ - └── §5 — Is the upstream backend healthy? - ├── GCP Secret Manager 5xx → §5a — escalate to platform - └── Vault unsealed / leader OK? → §5b — Vault outage (Layer 3) -``` - ---- - -## §1 — ESO controller health - -```bash -kubectl --context=$CTX -n external-secrets get pods -kubectl --context=$CTX -n external-secrets logs deploy/external-secrets --tail=100 | grep -iE 'error|failed|panic' -``` - -| Symptom | Cause | Layer | Action | -|---------|-------|-------|--------| -| Pods CrashLooping | Bad chart upgrade or RBAC misconfig | 1 | Check `helm-overrides//external-secrets/custom-values.yaml`. Last bump? Revert. | -| Pods Pending | Scheduling — wrong nodeSelector | 1 | See [pod-pending-scheduling.md](pod-pending-scheduling.md). | -| Pods Running, no logs about reconcile | Cluster-watch RBAC missing | 2 | Escalate. | -| `Forbidden` errors on CRD list | RBAC on the CRDs | 1 / 2 | Verify chart values' `rbac.create: true`. | - ---- - -## §2 — SecretStore / ClusterSecretStore status - -```bash -kubectl --context=$CTX get $KIND $STORE -o yaml | yq e '.status' -``` - -Look for `conditions[].status` and `conditions[].message`. - -| Status / message | Cause | Layer | Action | -|------------------|-------|-------|--------| -| `Ready: False, ValidationFailed` | Store spec invalid | 1 | Fix the `SecretStore` YAML in `helm-overrides//external-secrets/`. | -| `Ready: False, InvalidProviderConfig` | Provider block malformed | 1 | Validate `spec.provider.gcpsm.projectID` / `spec.provider.vault.server`. | -| `Ready: False, AuthFailed` (gcpsm) | Workload Identity binding broken | 2/3 | §4 below. | -| `Ready: False, AuthFailed` (vault) | KSA token review fails | 2/3 | §4 below. | -| `Ready: True` | Store OK; problem is elsewhere | — | Continue to §3. | - ---- - -## §3 — ExternalSecret spec validity - -```bash -kubectl --context=$CTX -n $NS get externalsecret $ES -o yaml | yq e '.spec.data, .spec.dataFrom' -``` - -For each `remoteRef.key`: - -- **GCP SM:** the key is the secret name in the project. Verify it exists: - ```bash - gcloud secrets list --project= --filter="name:" - ``` - (Read-only — no write to GCP SM from the agent.) - -- **Vault:** the key is the path under the engine. Cannot directly verify without Vault access; rely on the ESO controller's reconcile error message. - -If the controller logs say `secret not found in backend` → the `remoteRef.key` is wrong. **Layer 1 fix:** correct the key in the `ExternalSecret` YAML. - ---- - -## §4 — Auth path - -### §4a — GCP Secret Manager (Workload Identity) - -```bash -# The ServiceAccount the ESO controller (or this ExternalSecret's pod) runs as -kubectl --context=$CTX -n external-secrets get sa external-secrets -o yaml | yq e '.metadata.annotations' - -# Should have: -# iam.gke.io/gcp-service-account: @.iam.gserviceaccount.com -``` - -| Sub-check | Layer | Action | -|-----------|-------|--------| -| KSA missing the `iam.gke.io/gcp-service-account` annotation | 1 | Add via `helm-overrides//external-secrets/custom-values.yaml` `serviceAccount.annotations`. | -| GSA exists but no IAM binding to KSA | 2 | Escalate to platform — IAM is out of repo. | -| GSA lacks `roles/secretmanager.secretAccessor` on the project | 2 | Escalate to platform. | - -### §4b — Vault Kubernetes auth - -| Sub-check | Layer | Action | -|-----------|-------|--------| -| `SecretStore.spec.provider.vault.auth.kubernetes.role` references a Vault role | — | Read-only; verify against existing working stores on the same cluster. | -| ESO logs say `permission denied` from Vault | 3 | The role/policy on the Vault server is wrong. **Cannot fix from this repo.** Escalate to security / Vault platform team. | -| ESO logs say `Vault is sealed` | 3 | **Vault HA outage. Do not attempt to unseal.** Escalate immediately. | - ---- - -## §5 — Backend health - -### §5a — GCP Secret Manager - -GCP SM is a managed service. 5xx from it is rare and is a GCP-side incident. Action: escalate to platform; check GCP status dashboard. **No fix in this repo.** - -### §5b — Vault HA - -Vault HA on Meesho's fleet runs in production-only, no lower env. It is **Layer 3** — agent does not write to or operate Vault. Diagnostic *read* of pod state is OK; mutation is not. - -```bash -# Diagnostic only -kubectl --context=$CTX -n vault get pods -kubectl --context=$CTX -n vault logs | tail -50 -# Look for: "core: Vault is sealed", "leader election", "raft", panic stacks -``` - -| Observed | Action | -|----------|--------| -| Some Vault pods sealed (HA quorum still up) | Escalate to security team. **Do not unseal.** | -| All Vault pods sealed (full outage) | Page security team. ESO will surface stale data only as long as the in-memory cache holds. | -| Leader-election thrashing | Escalate; could be a network/raft issue. | -| Vault pod Pending | Scheduling — see [pod-pending-scheduling.md](pod-pending-scheduling.md). Even here, restart of a Vault pod requires a security-team-led unseal afterwards. | - ---- - -## Mitigations while Vault is down - -If a workload's `ExternalSecret` is failing because Vault is unavailable, options are limited: - -1. **Wait** — ESO caches the last-rendered Secret value. Pods that already mounted continue. New pod scheduling fails until Vault returns. -2. **Switch the `ExternalSecret` to GCP SM** if the same secret exists there (most do, with Vault as legacy). Layer 1 PR — change `secretStoreRef.name` to the GCP SM store. **Coordinate with security** before doing this in an outage. -3. **Hand-create the Secret as a temporary `kubectl apply`** — out of scope for the agent. This is incident response by a human operator. - -The agent should **not** auto-cut option 2 without explicit human approval — switching the source of truth for a secret has security implications. - ---- - -## Escalation matrix - -| Symptom | First responder | Escalate to | -|---------|-----------------|-------------| -| §1 (ESO down) | DevOps on-call | Platform team | -| §2 (Store config) | DevOps on-call | Layer 1 PR + reviewer | -| §3 (Bad remoteRef) | DevOps on-call | Layer 1 PR + reviewer | -| §4a (WI binding) | Platform team | — | -| §4b (Vault role/policy) | Security team | — | -| §5a (GCP SM 5xx) | Platform team | GCP support | -| §5b (Vault outage) | **Security team — page** | — | - ---- - -## Done conditions - -- `ExternalSecret` `status.conditions[type=Ready].status == True`. -- The downstream `Secret` exists with expected keys. -- Workload pods consuming the Secret are Running. -- ESO logs are quiet for at least 5 minutes after the fix. - ---- - -## What this repo can and cannot fix - -| Fix kind | Layer | This repo? | -|----------|-------|-----------| -| `ExternalSecret` / `SecretStore` YAML edits | 1 | Yes — values PR. | -| ESO chart values (sizing, RBAC, metrics) | 1 | Yes. | -| Workload-Identity KSA annotation | 1 | Yes (in chart values' `serviceAccount.annotations`). | -| GSA IAM bindings on GCP | 2 | No — platform team. | -| Vault server-side role/policy | 3 | **Never** — security team. | -| Vault unseal / raft / leader | 3 | **Never** — security team. | - ---- - -## Related - -- Schema: [../schemas/raw-manifest-sidecar-schema.md §ExternalSecret](../schemas/raw-manifest-sidecar-schema.md). -- Runbook: [pod-pending-scheduling.md](pod-pending-scheduling.md) — if any of the above pods are Pending. -- Runbook: [argocd-sync-failure.md](argocd-sync-failure.md) — if the `ExternalSecret` itself didn't sync. -- Boundaries: [../../global/AGENT_BOUNDARIES.md](../../global/AGENT_BOUNDARIES.md), [../../global/SANCTITY_RULES.md](../../global/SANCTITY_RULES.md). -- Escalation: [../../global/escalation-matrix.md](../../global/escalation-matrix.md). diff --git a/docs/platform/schemas/custom-values-schema.md b/docs/platform/schemas/custom-values-schema.md deleted file mode 100644 index 6c1b87c..0000000 --- a/docs/platform/schemas/custom-values-schema.md +++ /dev/null @@ -1,379 +0,0 @@ -# Schema — `custom-values.yaml` (Helm override) - -> Field-by-field annotation of the values file Argo CD's `valueFiles` references. -> Authoritative for `helm-overrides///custom-values.yaml`. -> -> Cross-reference: [coding-guidelines/helm-values.md](../../global/coding-guidelines/helm-values.md), [../../architecture.md](../../architecture.md). - ---- - -## What this file is — and isn't - -A `custom-values.yaml` is **not** a Helm chart. It is the *override* layer Argo CD merges over the chart's own `values.yaml` (or, for thin-wrapper charts, the upstream subchart's defaults). - -- **Schema is owned by the chart, not by us.** Every key here must be a key the chart understands. Adding a key the chart doesn't read does nothing. -- **No multi-doc YAML.** One document per file. -- **No `apiVersion` / `kind`.** This file is values, not a Kubernetes manifest. - ---- - -## Top-level shape (typical) - -```yaml -# image registry / repository / tag -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/ - tag: - pullPolicy: IfNotPresent - -# release-name pinning (rarely changed once set) -fullnameOverride: --prd # e.g. kube-state-metrics-dbc-dsci-prd - -# scaling -replicaCount: 3 # static, OR -autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 10 - targetCPUUtilizationPercentage: 70 - -# scheduling — bespoke per cluster -nodeSelector: - dedicated: # standard GKE - # cloud.google.com/compute-class: # GKE Autopilot -tolerations: - - key: dedicated - value: - effect: NoSchedule - -# resources -resources: - requests: - cpu: 250m - memory: 512Mi - limits: - cpu: 1 - memory: 2Gi - -# persistence (stateful charts only) -persistence: - enabled: true - storageClass: sc-pd-ssd # MUST exist in manifests/storageclass/ - size: 100Gi - accessModes: [ReadWriteOnce] - -# secrets — by reference, never inline -existingSecret: - -# probes -livenessProbe: - initialDelaySeconds: 30 - periodSeconds: 10 -readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 5 - -# ingress / Contour HTTPProxy (if applicable) -ingress: - enabled: false # most internal services - -# chart-specific top-level keys -# (varies — read the chart's values.yaml for the full schema) -``` - ---- - -## §image - -| Field | Required | Convention | Notes | -|-------|----------|------------|-------| -| `image.registry` | yes (production) | `asia-southeast1-docker.pkg.dev` | Meesho's Artifact Registry mirror. ([SANCTITY_RULES R11](../../global/SANCTITY_RULES.md)) | -| `image.repository` | yes | `meesho-devops-admin-0622/admin/sre/` | Mirror path. | -| `image.tag` | yes (production) | semver, datestamp, or git SHA | Never `latest`. Never unpinned. | -| `image.pullPolicy` | no | `IfNotPresent` | `Always` only for development; pulls slow rollout. | -| `image.pullSecrets` | no | omit | Mirror is public to fleet; pull secrets are a lock-in trap. | - -Note: chart authors disagree on the shape — some use `image: {repository, tag}` (no `registry`), some use a single `image: `, some split into `image.repository: /`. **Read the chart's own `values.yaml`** before structuring this block. - ---- - -## §release identity - -| Field | Required | Convention | Notes | -|-------|----------|------------|-------| -| `fullnameOverride` | varies | Set once at release creation; stable forever | Service DNS, PVC binding, ConfigMap refs depend on it. ([SANCTITY_RULES R9](../../global/SANCTITY_RULES.md)) | -| `nameOverride` | rare | Omit unless explicitly needed | | -| `commonLabels` / `commonAnnotations` | rare | Omit unless the chart documents the pattern | | - -For dataplane (`db-*`) clusters: `fullnameOverride: -dbc--prd` is the convention (e.g. `kube-state-metrics-dbc-dsci-prd`). For BU clusters, omit unless the chart's default name collides. - ---- - -## §scaling - -Charts diverge sharply here. Common shapes: - -```yaml -# Pattern A — static replicas -replicaCount: 3 - -# Pattern B — chart-level autoscaling block -autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 10 - targetCPUUtilizationPercentage: 70 - -# Pattern C — separate HPA resource (KEDA / VictoriaMetrics) -hpa: - enabled: true - min: 3 - max: 20 - metrics: - - type: Resource - resource: {name: cpu, target: {type: Utilization, averageUtilization: 70}} -``` - -| Rule | Why | -|------|-----| -| **Don't set `replicaCount`** when `autoscaling.enabled: true`. | HPA and the chart's static deployment fight; replica count flaps. | -| **Set both `minReplicas` and `maxReplicas`** when autoscaling. | Without `min`, the HPA can scale to zero on a quiet hour. | -| **`minReplicas` ≥ 2** for any production-traffic-path workload. | Single-replica releases die on node drain. | - ---- - -## §scheduling - -The single biggest source of silent mis-deploys. Per [SANCTITY_RULES R5](../../global/SANCTITY_RULES.md): - -| Cluster | Key style | -|---------|-----------| -| GKE Autopilot (`k8s-central-prd-ase1`, `k8s-dsgpu-prd-ase1`, `k8s-shared-int-ase1`) | `cloud.google.com/compute-class:` | -| Standard GKE | `dedicated:` | - -```yaml -# Standard GKE — Contour internal-0 on most clusters -nodeSelector: - dedicated: contour-internal-0 -tolerations: - - key: dedicated - value: contour-internal-0 - effect: NoSchedule - # often a second toleration to allow scheduling onto shared nodes - - key: dedicated - value: contour-shared - effect: NoSchedule - -# GKE Autopilot — same Contour internal-0 on k8s-central-prd-ase1 -nodeSelector: - cloud.google.com/compute-class: contour-internal-0-cc -tolerations: - - key: cloud.google.com/compute-class - value: contour-internal-0-cc - effect: NoSchedule - - key: cloud.google.com/compute-class - value: contour-shared-cc - effect: NoSchedule -``` - -The full per-cluster Contour matrix lives in [`contour-nodeselector-tolerations-summary.md`](../../../contour-nodeselector-tolerations-summary.md). For non-Contour apps, copy from a sibling app on the **same** cluster. - -| Field | Notes | -|-------|-------| -| `nodeSelector` | Hard requirement — no scheduling onto non-matching nodes. Use exactly one key (the cluster's pool key). | -| `tolerations` | Allow scheduling onto tainted nodes. May list multiple to allow shared nodes alongside dedicated. | -| `affinity.nodeAffinity` | Use sparingly; `nodeSelector` is sufficient for most cases here. | -| `topologySpreadConstraints` | Use for multi-zone resilience on chatty workloads (Contour, ingress-nginx). | - ---- - -## §resources - -```yaml -resources: - requests: - cpu: 250m - memory: 512Mi - limits: - cpu: 1 - memory: 2Gi -``` - -| Rule | Why | -|------|-----| -| **Always set `requests`** in production. | Without requests, pods are best-effort; first to evict on memory pressure. | -| **Set `limits` unless the chart explicitly recommends omitting them.** | Some sidecars (cert-manager, external-dns) deliberately skip `limits`. | -| **Don't copy resources from another cluster.** | Workload sizing is per-traffic-tier. Supply prd is not demand prd. | -| **CPU limits cause throttling**, not OOMKill. Memory limits cause OOMKill. Tune accordingly. | - ---- - -## §persistence - -```yaml -persistence: - enabled: true - storageClass: sc-pd-ssd - size: 100Gi - accessModes: [ReadWriteOnce] -``` - -Allowed `storageClass` values (must match a file under `manifests/storageclass/`): - -| Class | Backend | Use case | -|-------|---------|----------| -| `pd-standard-retain-dr` | GCP Persistent Disk standard, retention-on-delete | DR-critical state | -| `sc-pd-ssd` | GCP PD SSD | Default for write-heavy workloads (etcd, ClickHouse) | -| `sc-pd-standard` | GCP PD standard | Default for read-heavy / archival | -| `sc-filestore-standard` | GCP Filestore | Shared volumes (Jenkins build cache, JFrog filestore) | - -| Rule | Why | -|------|-----| -| **Never reference a `storageClass`** that doesn't exist in `manifests/storageclass/`. | PVC stays Pending forever. | -| **`size` must be set explicitly.** | Chart defaults are often wrong (8Gi for everything). | -| **`accessModes`** for `ReadWriteMany` requires Filestore. PD-backed classes are RWO. | - ---- - -## §secrets - -```yaml -# Chart-specific — read the chart's values.yaml -existingSecret: -# OR -auth: - existingSecret: -# OR -envFrom: - - secretRef: {name: } -``` - -| Rule | Why | -|------|-----| -| **Never inline `password:`, `apiKey:`, etc.** | Pre-commit TruffleHog will catch many; it won't catch all. ([SANCTITY_RULES R4](../../global/SANCTITY_RULES.md)) | -| **The `Secret` resource is materialised** by the cluster's `external-secrets` app. If the secret name doesn't appear in `helm-overrides//external-secrets/`, it doesn't exist. | -| **Reference, don't author.** Adding the secret material to a Helm-rendered `Secret` template defeats the External Secrets pattern. | - ---- - -## §probes - -```yaml -livenessProbe: - httpGet: {path: /healthz, port: http} - initialDelaySeconds: 30 - periodSeconds: 10 - timeoutSeconds: 3 - failureThreshold: 3 - -readinessProbe: - httpGet: {path: /ready, port: http} - initialDelaySeconds: 5 - periodSeconds: 5 - -# For slow-warm workloads (Jenkins, JFrog, ClickHouse) -startupProbe: - httpGet: {path: /healthz, port: http} - initialDelaySeconds: 60 - periodSeconds: 10 - failureThreshold: 60 # 10 minutes total runway -``` - -| Rule | Why | -|------|-----| -| **Always set `liveness` + `readiness`** for any long-running container. | -| **Use `startupProbe` for slow-warm workloads** instead of inflating `livenessProbe.initialDelaySeconds` to 600 s. | -| **Don't set the same probe to both `liveness` and `readiness`** — they have different purposes (kill vs gate). | - ---- - -## §ingress (Contour `HTTPProxy` / `Ingress`) - -For external-facing apps, ingress is via Contour `HTTPProxy`. Specifics depend on the cluster's Contour topology (multiple Contour instances per cluster — see [contour-nodeselector-tolerations-summary.md](../../../contour-nodeselector-tolerations-summary.md)). - -- The `external-dns` annotation is the DNS-binding step — match the cluster's DNS pattern. -- TLS terminates at Contour with `cert-manager`-issued certs; reference the issuer by name. -- Internal services use `contour-internal-0` / `contour-internal-1`; external services use `contour-external` / `contour-external-1`. The class is set via the `Service` annotation, not the values file directly. - ---- - -## End-to-end example — a typical observability-side override - -`helm-overrides/k8s-supply-prd-ase1/kube-state-metrics/custom-values.yaml`: - -```yaml -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - tag: v2.10.1 - -replicaCount: 2 - -resources: - requests: - cpu: 100m - memory: 256Mi - limits: - cpu: 500m - memory: 1Gi - -nodeSelector: - dedicated: monitoring -tolerations: - - key: dedicated - value: monitoring - effect: NoSchedule - -prometheus: - monitor: - enabled: true - honorLabels: true -``` - ---- - -## End-to-end example — a dataplane override - -`helm-overrides/db-2516183257845181-c-1204-195038-428/victoria-metrics-agent/custom-values.yaml`: - -```yaml -fullnameOverride: vmagent-dbc-supply-prd - -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/victoria-metrics-agent - tag: v1.96.0 - -replicaCount: 1 - -resources: - requests: - cpu: 100m - memory: 256Mi -``` - -(Dataplane clusters intentionally have minimal overrides.) - ---- - -## Validation - -```bash -# Render the chart locally with this values file -helm template helm-templates/ \ - -f helm-overrides///custom-values.yaml | head -80 - -# Dry-run diff against the live cluster (requires kubectl context + helm-diff plugin) -helm diff upgrade helm-templates/ \ - -f helm-overrides///custom-values.yaml - -# Lint -yamllint helm-overrides///custom-values.yaml -helm lint helm-templates/ - -# Sanity: storageClass referenced exists -yq e '.persistence.storageClass' helm-overrides///custom-values.yaml \ - | xargs -I{} ls manifests/storageclass/{}.yaml 2>/dev/null \ - || echo "WARN: storageClass not in manifests/storageclass/" -``` diff --git a/docs/platform/schemas/incubator-values-schema.md b/docs/platform/schemas/incubator-values-schema.md deleted file mode 100644 index 10caf55..0000000 --- a/docs/platform/schemas/incubator-values-schema.md +++ /dev/null @@ -1,195 +0,0 @@ -> Per AI Blitz Plan §platform.schemas. Layer: 1. Repo: devops-infra-helm-charts. - -# Schema — Incubator-tool values + sidecar contract - -> **Scope:** the values-side contract for incubator (newly-vendored) infrastructure tools that live under `helm-overrides///`. -> -> **Out of scope (explicit):** the Argo CD `Application` / `ApplicationSet` manifest that registers the incubator tool with a cluster's Argo. Those manifests live in the **sister repo** `github.com/Meesho/devops-infra-argo-config`, NOT here. This file documents only what `devops-infra-helm-charts` is responsible for: the values + raw sidecar manifests Argo CD reads from this repo. - -This schema complements [custom-values-schema.md](custom-values-schema.md) (general values shape) and [raw-manifest-sidecar-schema.md](raw-manifest-sidecar-schema.md) (sidecar manifest shapes). Read both first. - ---- - -## What "incubator" means here - -A tool is **incubator** while it is being trialled on one or two clusters before fleet-wide rollout. In this repo that maps to: - -- A new chart directory under `helm-templates//` (often a thin wrapper `Chart.yaml` with an upstream dep). -- An override under `helm-overrides///` for the trial cluster(s) only — typically `k8s-shared-int-ase1` (integration / pre-prod) first, sometimes one BU prod cluster as a canary. -- Possibly raw sidecar manifests alongside the values file. - -Once the tool graduates from incubator status, the override pattern is identical to other infra tools — the "incubator" label is operational, not structural. This schema codifies the conventions that keep early-stage adoption sane. - ---- - -## Directory layout - -``` -helm-overrides/// -├── custom-values.yaml # Helm values (required if chart is Helm-based) -├── computeclass/ # Optional — GKE Autopilot ComputeClass -│ └── -cc.yaml -├── external-dns-services/ # Optional — Service for external-dns annotation -│ └── .yaml -├── external-secrets/ # Optional — ExternalSecret resources -│ └── .yaml -└── .yaml # Tool-specific raw manifest (CRD, ConfigMap) -``` - -The Argo `Application` for this directory (sister repo) determines whether files are Helm-rendered, raw-applied, or layered (see [raw-manifest-sidecar-schema.md §Layered-with-Helm vs standalone](raw-manifest-sidecar-schema.md)). - ---- - -## `custom-values.yaml` keys an incubator chart should expose - -A chart promoted to incubator status MUST surface (i.e. let the override file set without forking templates) at minimum: - -| Key | Why required | -|-----|--------------| -| `image.registry`, `image.repository`, `image.tag` | Production overrides pin to `asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/` (Meesho Artifact Registry mirror, not Docker Hub). | -| `image.pullPolicy` | Default `IfNotPresent`. | -| `resources.requests.cpu`, `.memory` | No defaults assumed; Autopilot scheduling requires explicit requests. | -| `resources.limits.cpu`, `.memory` | Same. | -| `nodeSelector` | Per-cluster scheduling (see [custom-values-schema.md](custom-values-schema.md)). | -| `tolerations` | Per-cluster taints. | -| `affinity` (optional) | Pod anti-affinity for replicated tools. | -| `serviceAccount.create`, `.name`, `.annotations` | Workload Identity — `iam.gke.io/gcp-service-account` annotation goes here. | -| `replicaCount` (or `autoscaling.{enabled,minReplicas,maxReplicas}`) | Sizing. | -| `persistence.enabled`, `.storageClass`, `.size` | If stateful. `storageClass` MUST exist in `manifests/storageclass/`. See [storageclass-priorityclass-schema.md](storageclass-priorityclass-schema.md). | -| `priorityClassName` | If the tool needs preemption priority — must reference a class in `manifests/priorityclass//`. | -| `podLabels`, `podAnnotations` | For prometheus.io scrape annotations and team-attribution labels. | -| `extraEnv` (or `env`) | For environment-specific knobs the chart's templates don't already accept. | - -If the upstream chart doesn't expose these — that's a chart-bug. **Do not** fork the chart in `helm-templates/` to add them ([NEVER-DO list](../../../CLAUDE.md)). Either upstream-PR the chart or wrap with a small Meesho-owned chart that re-exposes the keys. - ---- - -## Values-file conventions for incubator tools - -1. **Pin every image tag.** Never `latest`, never an unpinned SHA. This is the same rule as production overrides; incubator status does not relax it. -2. **Set explicit `nodeSelector` and `tolerations`** authored from scratch using sibling apps on the same cluster. Cross-cluster copying is forbidden — see [custom-values-schema.md](custom-values-schema.md). -3. **Use `fullnameOverride` deliberately or not at all.** Once set, do not change ([NEVER-DO list](../../../CLAUDE.md)). Incubator graduations to other clusters should reuse the same `fullnameOverride` string for portability. -4. **Don't enable `autoscaling`** in the first incubator deploy. Pin `replicaCount: 1` (or 2 for HA-mandatory) until you have a load profile. -5. **Don't expose the tool externally** in the incubator phase. No `external-dns-services/` until it's promoted to a cluster's stable inventory. -6. **Annotate the values file** with a top-of-file YAML comment: `# Incubator: cluster=<...>, owner=<...>, graduation-target=`. Comment is plain text; not parsed; serves as reviewer signal. -7. **Confine secrets to `ExternalSecret`** under `external-secrets/`. No inline `existingSecret:` referencing a hand-applied Secret. - ---- - -## Raw sidecar manifests in incubator directories - -Concrete shapes already documented elsewhere; this section calls out which ones routinely appear with incubators. - -### `computeclass/*-cc.yaml` (GKE Autopilot only) - -Required when the override's `nodeSelector` uses `cloud.google.com/compute-class: ` and that class doesn't already exist on the cluster. The `metadata.name` MUST equal the value referenced. See [raw-manifest-sidecar-schema.md §ComputeClass](raw-manifest-sidecar-schema.md). - -Example incubator pattern: - -```yaml -apiVersion: autoscaling.gke.io/v1 -kind: ComputeClass -metadata: - name: -cc - namespace: -spec: - priorities: - - machineFamily: c4d - minCores: 2 - minMemory: 8Gi - nodePoolAutoCreation: - enabled: true -``` - -Per-cluster only — never copy between clusters. - -### `external-dns-services/*.yaml` - -Skip in the first incubator deploy. Only add once the tool has a stable internal-only DNS need. See [raw-manifest-sidecar-schema.md §Service for external-dns binding](raw-manifest-sidecar-schema.md). - -### `external-secrets/*.yaml` - -Always present if the tool needs secrets. Use the existing per-cluster `SecretStore` / `ClusterSecretStore`; do not author new stores in an incubator PR. See [raw-manifest-sidecar-schema.md §ExternalSecret](raw-manifest-sidecar-schema.md). - -### `elastic-cluster/argo-launch.yaml`-style operator-managed CRD launches - -If the incubator tool is an operator (e.g. ECK, Pyroscope), a sidecar CRD instance often lives alongside the operator's chart values: - -``` -helm-overrides/// -├── custom-values.yaml # operator chart values -└── / - └── argo-launch.yaml # the actual workload CRD instance -``` - -Argo CD applies both in one Application (directory loader). The CRD instance must: - -- Reference an `apiVersion` whose CRD the operator has already installed. -- Pin its own image / version explicitly. -- Avoid hand-rolling values that the operator templates would otherwise compute. - ---- - -## Argo CD — sister repo contract (for cross-reference only) - -The Application that points at this directory lives in `github.com/Meesho/devops-infra-argo-config`. For incubator tools the Application typically: - -- Has `syncPolicy.automated` **disabled** (manual sync — incubator-grade safety; see [argocd.md](../../global/coding-guidelines/argocd.md)). -- Is named `-` to be unambiguous. -- Lives in the cluster's namespace under `apps//`. - -This file does NOT instruct on authoring the Application — that PR is in the sister repo. Cross-link the sister-repo PR in the values-side PR description. - ---- - -## Validation - -Same as any Helm override. Before opening a PR: - -```bash -yamllint helm-overrides///custom-values.yaml - -helm template helm-templates/ \ - -f helm-overrides///custom-values.yaml > /tmp/render.yaml - -# Validate any sidecar manifests against the cluster's CRDs -kubectl --context= --dry-run=server \ - -f helm-overrides///.yaml apply -``` - -For tool-specific CRDs that aren't in vanilla Kubernetes, prefer `--dry-run=server` (the cluster validates against the registered CRD) over `--dry-run=client` (only client-side schema, often misses required fields). - ---- - -## Graduation: incubator → stable - -When the tool has been stable for some period (usually 2–4 weeks) and is ready to fleet-roll: - -1. Remove the top-of-file `# Incubator:` comment. -2. Open new override directories in target clusters — author from scratch each time, do not copy. -3. Open the matching sister-repo `Application` PRs (one per new cluster) or extend the `ApplicationSet`. -4. If the chart was a thin wrapper, audit `Chart.yaml` `dependencies[].version` is current. -5. Promote the chart's documentation in `helm-templates//README.md` (if a fork was needed) or note in PR description that no fork was needed. - ---- - -## Anti-patterns - -1. **Forking the chart in `helm-templates//templates/`** to add a missing values key. Wrap or upstream-PR; don't fork. See NEVER-DO. -2. **Copying the entire incubator directory across clusters** for graduation. Per-cluster scheduling is bespoke. -3. **Enabling autoscaling on day one.** No load profile = thrashy autoscaler. -4. **Promoting from `k8s-shared-int-ase1` straight to a critical BU cluster.** Insert a low-traffic prod canary first. -5. **Inlining secrets** "just for the trial." TruffleHog will block; even if it didn't, the leak is real. -6. **Setting `fullnameOverride`** without a graduation plan. The string travels — bad ones travel forever. - ---- - -## Related - -- Schema: [custom-values-schema.md](custom-values-schema.md) — general values shape. -- Schema: [raw-manifest-sidecar-schema.md](raw-manifest-sidecar-schema.md) — sidecar shapes. -- Schema: [storageclass-priorityclass-schema.md](storageclass-priorityclass-schema.md) — cluster-singleton resources. -- Procedure: [../procedures/onboard-app-to-cluster.md](../procedures/onboard-app-to-cluster.md) — the procedure used to land an incubator override. -- Procedure: [../procedures/fork-upstream-chart.md](../procedures/fork-upstream-chart.md) — only when a fork is genuinely required. -- Coding guideline: [../../global/coding-guidelines/helm-values.md](../../global/coding-guidelines/helm-values.md). -- Coding guideline: [../../global/coding-guidelines/argocd.md](../../global/coding-guidelines/argocd.md) — sister-repo Application conventions. diff --git a/docs/platform/schemas/raw-manifest-sidecar-schema.md b/docs/platform/schemas/raw-manifest-sidecar-schema.md deleted file mode 100644 index 2da9d2e..0000000 --- a/docs/platform/schemas/raw-manifest-sidecar-schema.md +++ /dev/null @@ -1,191 +0,0 @@ -# Schema — Raw-manifest sidecars in `helm-overrides/` - -> Field-by-field guidance for `helm-overrides///.yaml` — Kubernetes manifests applied **alongside** a Helm release, not through it. - ---- - -## What this is - -Some Argo CD Applications point at a directory containing both a Helm `custom-values.yaml` *and* one or more raw Kubernetes manifests. The raw manifests are not Helm-rendered; Argo CD applies them as-is. - -Common shapes seen in this repo: - -| Path pattern | What it is | -|--------------|------------| -| `helm-overrides///computeclass/*-cc.yaml` | GKE Autopilot `ComputeClass` resource — declares a node-pool/compute-class profile referenced by `nodeSelector` in the Helm values. | -| `helm-overrides///external-dns-services/.yaml` | A `Service` that exists purely to carry an `external-dns.alpha.kubernetes.io/hostname` annotation, binding a DNS name to a workload. | -| `helm-overrides//elastic-cluster/argo-launch.yaml` | An `ElasticCluster` (ECK CRD) launched alongside the operator. | -| `helm-overrides///mimir-distributed/alertmanager_config.yaml` | Inlined Alertmanager config materialised as a `ConfigMap`. | - -If the sister-repo `Application` for this directory has `path: helm-overrides///`, then **every YAML file in the directory is applied** — Argo CD's directory loader treats them as a single deployment unit. - ---- - -## File-level conventions - -| Rule | Why | -|------|-----| -| **One Kubernetes resource per file** unless they're tightly coupled (e.g. a `Service` + a `ServiceAccount` referenced by it). | Reviewer cognition; rollback granularity. | -| **Pin `apiVersion` explicitly.** | `extensions/v1beta1` and `networking.k8s.io/v1beta1` are real footguns — both are gone in current Kubernetes. | -| **Set `metadata.namespace` explicitly.** | Don't rely on the Argo `Application.spec.destination.namespace` carrying through — different cluster Argos behave differently here. | -| **Trailing newline at EOF.** | -| **No multi-doc (`---`) within one file** unless genuinely required. | - ---- - -## Top-level shape - -```yaml -apiVersion: # e.g. v1, networking.k8s.io/v1, autoscaling.gke.io/v1 -kind: # e.g. Service, ComputeClass, ConfigMap, ExternalSecret -metadata: - name: - namespace: - labels: {...} # optional - annotations: {...} # often the load-bearing field (DNS, LB) -spec: - ... # kind-specific -``` - ---- - -## Common kinds in this repo - -### `ComputeClass` (`autoscaling.gke.io/v1`) - -GKE Autopilot uses `ComputeClass` resources to declare named node profiles. The Helm values reference them via `cloud.google.com/compute-class:` keys. - -```yaml -apiVersion: autoscaling.gke.io/v1 -kind: ComputeClass -metadata: - name: contour-internal-0-cc - namespace: projectcontour -spec: - priorities: - - machineFamily: c4d - minCores: 4 - minMemory: 16Gi - nodePoolAutoCreation: - enabled: true -``` - -| Rule | Why | -|------|-----| -| **`metadata.name` MUST match the value referenced by `nodeSelector` in the Helm values.** | Otherwise pods stay `Pending`. | -| **Per-cluster.** A `ComputeClass` for `k8s-central-prd-ase1` does not exist on `k8s-supply-prd-ase1`. Don't share files. | -| **The matrix of which `ComputeClass` exists where** is recorded in [contour-nodeselector-tolerations-summary.md](../../../contour-nodeselector-tolerations-summary.md). | - -### `Service` for `external-dns` binding - -```yaml -apiVersion: v1 -kind: Service -metadata: - name: -dns - namespace: - annotations: - external-dns.alpha.kubernetes.io/hostname: .meeshogcp.in - cloud.google.com/load-balancer-type: Internal -spec: - type: ClusterIP - selector: {app.kubernetes.io/name: } - ports: - - port: 80 - targetPort: 8080 -``` - -| Rule | Why | -|------|-----| -| **`metadata.annotations.external-dns.alpha.kubernetes.io/hostname`** is the contract. Check the cluster's existing DNS pattern before authoring. | -| **`selector` must match labels of an actual workload Pod** in the same namespace. | -| **Cloud-LB annotations** (`cloud.google.com/load-balancer-type`, etc.) — copy from a sibling on the same cluster. | - -### `ExternalSecret` (per-cluster `external-secrets/`) - -```yaml -apiVersion: external-secrets.io/v1beta1 -kind: ExternalSecret -metadata: - name: - namespace: -spec: - refreshInterval: 1h - secretStoreRef: - name: gcp-sm-store # cluster-level SecretStore, defined elsewhere - kind: ClusterSecretStore - target: - name: - creationPolicy: Owner - data: - - secretKey: - remoteRef: - key: - version: latest -``` - -| Rule | Why | -|------|-----| -| **`spec.target.name` is the `Secret` name** the workload's `existingSecret:` references. | -| **`spec.secretStoreRef.name`** must reference a `SecretStore` / `ClusterSecretStore` that already exists on the cluster. | -| **`creationPolicy: Owner`** is the default; `creationPolicy: Merge` is for the rare case where another tool also manages the same `Secret`. | - -### `ConfigMap` for inlined config - -```yaml -apiVersion: v1 -kind: ConfigMap -metadata: - name: - namespace: -data: - : | - -``` - -| Rule | Why | -|------|-----| -| **Use `|` (literal block scalar)** for multi-line content with significant whitespace. | -| **Don't inline secrets** (TruffleHog will catch obvious ones; subtle ones slip through). | -| **Reference from the workload's chart values** via the `ConfigMap` name; don't duplicate config across files. | - -### `ElasticCluster` / other CRDs - -For ECK, Pyroscope, etc. — the schema is the operator's, not Kubernetes's. **Read the operator's CRD docs** before authoring; copy a sibling cluster's existing `argo-launch.yaml` first. - ---- - -## Layered-with-Helm vs standalone - -| Pattern | Argo Application points at | -|---------|----------------------------| -| Pure Helm | `path: helm-overrides///` with `helm.valueFiles: ['custom-values.yaml']` | -| Pure raw manifests | `path: helm-overrides///` with `directory.recurse: true`, no `helm:` block | -| Layered | `path: helm-overrides///` with `helm.valueFiles: ['custom-values.yaml']` AND extra files alongside | - -Argo CD's behaviour for layered directories depends on its `directory.include` / `directory.exclude` settings — when in doubt, read the sister-repo `Application` to see exactly what gets picked up. - ---- - -## Validation - -```bash -# Validate the manifest against its API -kubectl --dry-run=client -f helm-overrides///.yaml apply - -# Lint -yamllint helm-overrides///.yaml - -# Confirm the Argo Application loads this path -# (read the matching file in github.com/Meesho/devops-infra-argo-config) -``` - ---- - -## Anti-patterns - -1. **Inlining secrets** in a `ConfigMap` "to ship a hotfix." It's still a secret. Use `ExternalSecret`. -2. **Using `apiVersion: v1beta1`** of a CRD whose stable version exists. Pin to the highest stable. -3. **A `Service` whose `selector` doesn't match any Pod** — silent: the Service exists, DNS resolves, no endpoints. -4. **Cross-cluster cloning** of a `ComputeClass` or DNS `Service` without rewriting cluster-specific fields. -5. **Hand-rendered Helm output** dropped into a sidecar file (a release "freeze"). The chart bumps and your snapshot rots. diff --git a/docs/platform/schemas/storageclass-priorityclass-schema.md b/docs/platform/schemas/storageclass-priorityclass-schema.md deleted file mode 100644 index cc8e5ee..0000000 --- a/docs/platform/schemas/storageclass-priorityclass-schema.md +++ /dev/null @@ -1,158 +0,0 @@ -# Schema — `manifests/storageclass/` and `manifests/priorityclass/` - -> Cluster-wide singletons. **High blast radius — every PVC / scheduling decision in the cluster is affected.** -> -> Cross-reference: [SANCTITY_RULES R10](../../global/SANCTITY_RULES.md), [coding-guidelines/helm-values.md §persistence](../../global/coding-guidelines/helm-values.md). - ---- - -## §StorageClass (`manifests/storageclass/*.yaml`) - -These are **repo-global** — one StorageClass file is applied to every cluster that consumes it. A wrong reclaim policy or volume-binding mode breaks every new PVC. - -### Inventory - -| File | Backend | Reclaim | Binding | Use case | -|------|---------|---------|---------|----------| -| `pd-standard-retain-dr.yaml` | GCP PD standard | `Retain` | `WaitForFirstConsumer` | DR-critical state — do not lose data on PVC delete | -| `sc-pd-ssd.yaml` | GCP PD SSD | `Delete` | `WaitForFirstConsumer` | Default for write-heavy (etcd, ClickHouse) | -| `sc-pd-standard.yaml` | GCP PD standard | `Delete` | `WaitForFirstConsumer` | Default for read-heavy / archival | -| `sc-filestore-standard.yaml` | GCP Filestore | varies | varies | Shared `ReadWriteMany` volumes (Jenkins build cache, JFrog filestore) | - -### Schema - -```yaml -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - name: sc-pd-ssd # MUST match filename - annotations: - storageclass.kubernetes.io/is-default-class: "false" # exactly one default per cluster -provisioner: pd.csi.storage.gke.io -parameters: - type: pd-ssd -reclaimPolicy: Delete # Delete | Retain -volumeBindingMode: WaitForFirstConsumer -allowVolumeExpansion: true -``` - -| Field | Convention | Notes | -|-------|------------|-------| -| `metadata.name` | matches filename (without `.yaml`) | Helm values reference by name. | -| `metadata.annotations."storageclass.kubernetes.io/is-default-class"` | `"false"` for all of these | Exactly one StorageClass should be `"true"` per cluster (declared elsewhere, often by Terraform/cluster-bootstrap). | -| `provisioner` | `pd.csi.storage.gke.io` (PD) or `filestore.csi.storage.gke.io` (Filestore) | GCP CSI drivers. | -| `parameters.type` | `pd-ssd`, `pd-standard`, `pd-balanced` | Cost/perf trade-off. | -| `reclaimPolicy` | `Delete` (default) or `Retain` (DR) | **Changing this on an existing StorageClass does not retroactively change PVCs.** | -| `volumeBindingMode` | `WaitForFirstConsumer` | Avoids zone-mismatch on multi-zone clusters. `Immediate` only for `ReadWriteMany`. | -| `allowVolumeExpansion` | `true` | Required for online resizing. Default false on some classes — set explicitly. | - -### Hard rules - -1. **Never delete a StorageClass referenced by an existing PVC.** New PVCs against the deleted class fail; existing bound PVCs survive but lose the ability to expand. -2. **Never change `reclaimPolicy` from `Delete` → `Retain`** as a "safety improvement" without auditing every PVC. Existing bound PVs keep their old policy; new PVs get the new policy. Drift. -3. **Never change `provisioner`** — that's a delete-and-recreate, not an edit. Existing PVs become orphaned. -4. **Never make a new StorageClass the cluster default** without coordinating with the platform team. A wrong default class hijacks every PVC that doesn't pin a class explicitly. -5. **Validation:** every StorageClass referenced by any `helm-overrides/*/persistence.storageClass:` must have a corresponding file here. - -```bash -# Find every PVC reference -grep -rE 'storageClass(Name)?:' helm-overrides | sort -u - -# Find every StorageClass file -ls manifests/storageclass/*.yaml -``` - ---- - -## §PriorityClass (`manifests/priorityclass//*.yaml`) - -**Per-cluster** but **cluster-wide** — affects scheduling priority for every pod that references the class. - -### Inventory pattern - -``` -manifests/priorityclass/ - k8s-central-prd-ase1/ - priorityclass-high.yaml - priorityclass-low.yaml - k8s-supply-prd-ase1/ - priorityclass-high.yaml - priorityclass-low.yaml - … -``` - -Most BU clusters get a `high` and a `low` class. Specialty clusters (`mqkafka`, `dsgpu`, dataplane `db-*`) sometimes have additional tiers. - -### Schema - -```yaml -apiVersion: scheduling.k8s.io/v1 -kind: PriorityClass -metadata: - name: high-priority -value: 1000000 # higher = more important -globalDefault: false # NEVER true on these — would clash with system defaults -description: "High-priority workloads — promoted ahead of general pods on resource pressure" -preemptionPolicy: PreemptLowerPriority # default; alternative is Never -``` - -| Field | Convention | Notes | -|-------|------------|-------| -| `metadata.name` | `high-priority`, `low-priority`, etc. | Workload pods reference by name. | -| `value` | High: 1,000,000 / Low: 100 / Critical (rare): >1,000,000,000 | Kubernetes system pods reserve `> 2,000,000,000`; don't conflict. | -| `globalDefault` | **always `false`** | A `true` here applies to every pod that doesn't pin a class — chaos. | -| `description` | one-line | For audit log. | -| `preemptionPolicy` | `PreemptLowerPriority` (default) or `Never` | `Never` for batch workloads that shouldn't kick others out. | - -### Hard rules - -1. **Never set `globalDefault: true`** on any PriorityClass here. The cluster's implicit default is what we want. -2. **Never raise `value`** of an existing class without auditing what gets preempted. A bump from 1,000,000 → 10,000,000 changes which pods get evicted under pressure. -3. **Never delete a PriorityClass** referenced by any workload — pods that referenced it become invalid. -4. **The same `name` must mean the same thing across clusters.** `high-priority` on supply ≠ `high-priority` on demand at the *value* level is an audit nightmare. Keep values consistent. -5. **Validation:** find every pod-spec reference: - ```bash - grep -rE 'priorityClassName:' helm-overrides - ``` - ---- - -## §The "PV/PVC singletons" — Jenkins / JFrog filestore - -Beyond StorageClass and PriorityClass, `manifests/` also holds **per-env, one-shot** PV/PVC pairs: - -``` -manifests/jenkins-filestore-caching/ - dev/{pv,pvc}.yaml - prd/{pv,pvc}.yaml -manifests/jenkins-gcs-caching/{pv,pvc,sc-gcs}.yaml -manifests/jfrog-filestore-data/ - dev/... - prd/... -``` - -These are **already-created PVs being adopted** — typically because the underlying disk (Filestore mount, GCS bucket) was provisioned by Terraform or by hand. The PV file binds the existing disk to the cluster; the PVC file binds a workload to the PV. - -| Rule | Why | -|------|-----| -| **Don't change `spec.csi.volumeHandle` / `spec.gcePersistentDisk.pdName`** without coordinating with Terraform. The handle is the disk identity. | -| **Don't change `spec.persistentVolumeReclaimPolicy`** from `Retain` to `Delete` on these. The disk has data on it. | -| **Don't move dev→prd or vice versa** — they reference different physical disks. | -| **Treat as platform-team review.** Any change here is a 1:1 disk operation. | - ---- - -## Validation script - -```bash -# StorageClass: every name in helm-overrides has a file here -for sc in $(grep -rhE 'storageClass(Name)?:' helm-overrides \ - | sed -E 's/.*storageClass(Name)?:\s*//' \ - | tr -d '"' | sort -u); do - [ -f "manifests/storageclass/${sc}.yaml" ] || echo "MISSING: $sc" -done - -# PriorityClass: every name referenced has a file in the matching cluster -grep -rE 'priorityClassName:' helm-overrides -# (cross-reference manually against manifests/priorityclass//) -``` diff --git a/helm-overrides/db-2516183257845181-c-1204-195038-428/kube-state-metrics/custom-values.yaml b/helm-overrides/db-2516183257845181-c-1204-195038-428/kube-state-metrics/custom-values.yaml deleted file mode 100644 index b614e22..0000000 --- a/helm-overrides/db-2516183257845181-c-1204-195038-428/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-dbc-dsci-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-dbc-dsci-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dbc-dsci" - team: "dbc-dsci-sre" - service: "kube-state-metrics-dbc-dsci-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "devops" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 10m - memory: 50Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/db-2516183257845181-c-1204-195038-428/victoria-metrics-agent/custom-values.yaml b/helm-overrides/db-2516183257845181-c-1204-195038-428/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index c781725..0000000 --- a/helm-overrides/db-2516183257845181-c-1204-195038-428/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,296 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-dbc-dsci-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-dbc-dsci-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dbc-desre-vmagent-prd@meesho-dbc-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-dbc.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-dbc-prd.meeshogcp.in/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dbc-dsci" - team: "dbc-dsci-sre" - service: "vmagent-dbc-dsci-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dbc-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dbc-dsci-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 2 - memory: 4Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "devops" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/db-3474993620361343-3-1215-134240-999/ingress-nginx-internal/custom-values.yaml b/helm-overrides/db-3474993620361343-3-1215-134240-999/ingress-nginx-internal/custom-values.yaml deleted file mode 100644 index c3f5b9c..0000000 --- a/helm-overrides/db-3474993620361343-3-1215-134240-999/ingress-nginx-internal/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 6 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dbc-internal-prd"}}}' - nodeSelector: - dedicated: devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" diff --git a/helm-overrides/db-3474993620361343-3-1215-134240-999/kube-state-metrics/custom-values.yaml b/helm-overrides/db-3474993620361343-3-1215-134240-999/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 77fd900..0000000 --- a/helm-overrides/db-3474993620361343-3-1215-134240-999/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-dbc-dengg-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-dbc-dengg-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dbc-dengg" - team: "dbc-dengg-sre" - service: "kube-state-metrics-dbc-dengg-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "devops" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 10m - memory: 50Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/db-3474993620361343-3-1215-134240-999/victoria-metrics-agent/custom-values.yaml b/helm-overrides/db-3474993620361343-3-1215-134240-999/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index a31533a..0000000 --- a/helm-overrides/db-3474993620361343-3-1215-134240-999/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,296 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-dbc-dengg-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-dbc-dengg-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dbc-desre-vmagent-prd@meesho-dbc-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-dbc.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-dbc-prd.meeshogcp.in/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dbc-dengg" - team: "dbc-dengg-sre" - service: "vmagent-dbc-dengg-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dbc-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dbc-dengg-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 2 - memory: 4Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "devops" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/db-3474993620361343-3-1215-134240-999/victoria-metrics-insert/custom-values.yaml b/helm-overrides/db-3474993620361343-3-1215-134240-999/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index c72b885..0000000 --- a/helm-overrides/db-3474993620361343-3-1215-134240-999/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,234 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dbc-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-dbc-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "dbc" - team: "sre" - service: "vminsert-dbc-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 2 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmcommon" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmcommon" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 4 - memory: 2Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-dbc-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: nginx-internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/db-3474993620361343-3-1215-134240-999/victoria-metrics-select/custom-values.yaml b/helm-overrides/db-3474993620361343-3-1215-134240-999/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index 95e0555..0000000 --- a/helm-overrides/db-3474993620361343-3-1215-134240-999/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,291 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dbc-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-dbc-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 180s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "dbc" - team: "dbc-sre" - service: "vmselect-dbc-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 15 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - nodeSelector: - dedicated: "vmselect" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 5 - memory: 6Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-dbc-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/db-3474993620361343-3-1215-134240-999/victoria-metrics-storage/custom-values.yaml b/helm-overrides/db-3474993620361343-3-1215-134240-999/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index 17393e0..0000000 --- a/helm-overrides/db-3474993620361343-3-1215-134240-999/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,316 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-dbc-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: sc-pd-ssd - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 300Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "dbc" - team: "sre" - service: "vmstorage-dbc-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 5 - memory: 40Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.93.7-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-dbc-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/db-5070361081051941-0-0108-201559-719/kube-state-metrics/custom-values.yaml b/helm-overrides/db-5070361081051941-0-0108-201559-719/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 6772756..0000000 --- a/helm-overrides/db-5070361081051941-0-0108-201559-719/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-dbc-backend-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-dbc-backend-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dbc-backend" - team: "dbc-backend-sre" - service: "kube-state-metrics-dbc-backend-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "devops" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 10m - memory: 50Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/db-5070361081051941-0-0108-201559-719/victoria-metrics-agent/custom-values.yaml b/helm-overrides/db-5070361081051941-0-0108-201559-719/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index 2848963..0000000 --- a/helm-overrides/db-5070361081051941-0-0108-201559-719/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,296 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-dbc-backend-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-dbc-backend-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dbc-desre-vmagent-prd@meesho-dbc-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-dbc.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-dbc-prd.meeshogcp.in/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dbc-backend" - team: "dbc-backend-sre" - service: "vmagent-dbc-backend-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dbc-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dbc-backend-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 2 - memory: 4Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "devops" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/db-8070720513218850-d-1128-192026-437/kube-state-metrics/custom-values.yaml b/helm-overrides/db-8070720513218850-d-1128-192026-437/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 116bfb4..0000000 --- a/helm-overrides/db-8070720513218850-d-1128-192026-437/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-dbc-dping-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-dbc-dping-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dbc-dping" - team: "dbc-dping-sre" - service: "kube-state-metrics-dbc-dping-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "devops" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 10m - memory: 50Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/db-8070720513218850-d-1128-192026-437/victoria-metrics-agent/custom-values.yaml b/helm-overrides/db-8070720513218850-d-1128-192026-437/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index 3065034..0000000 --- a/helm-overrides/db-8070720513218850-d-1128-192026-437/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,296 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-dbc-dping-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-dbc-dping-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dbc-desre-vmagent-prd@meesho-dbc-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-dbc.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-dbc-prd.meeshogcp.in/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dbc-dping" - team: "dbc-dping-sre" - service: "vmagent-dbc-dping-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dbc-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dbc-dping-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 2 - memory: 4Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "devops" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/db-8405596239050069-0-0222-204522-227/kube-state-metrics/custom-values.yaml b/helm-overrides/db-8405596239050069-0-0222-204522-227/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 6b8d096..0000000 --- a/helm-overrides/db-8405596239050069-0-0222-204522-227/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-dbc-dpcon-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-dbc-dpcon-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dbc-dpcon" - team: "dbc-dpcon-sre" - service: "kube-state-metrics-dbc-dpcon-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "devops" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 10m - memory: 50Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/db-8405596239050069-0-0222-204522-227/victoria-metrics-agent/custom-values.yaml b/helm-overrides/db-8405596239050069-0-0222-204522-227/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index a2e4b9c..0000000 --- a/helm-overrides/db-8405596239050069-0-0222-204522-227/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,296 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-dbc-dpcon-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-dbc-dpcon-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dbc-desre-vmagent-prd@meesho-dbc-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-dbc.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-dbc-prd.meeshogcp.in/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dbc-dpcon" - team: "dbc-dpcon-sre" - service: "vmagent-dbc-dpcon-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dbc-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dbc-dpcon-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 2 - memory: 4Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "devops" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/gke-central-prd-ase1a/README.md b/helm-overrides/gke-central-prd-ase1a/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/gke-central-prd-ase1a/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/gke-central-prd-ase1a/ai-gateway-ext/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/ai-gateway-ext/custom-values.yaml deleted file mode 100644 index 7938302..0000000 --- a/helm-overrides/gke-central-prd-ase1a/ai-gateway-ext/custom-values.yaml +++ /dev/null @@ -1,295 +0,0 @@ -# Custom values for Bifrost (ai-gateway-ext) - Meesho Production -# Usage: helm install bifrost ./helm-templates/bifrost-v1.5.12-latest/ -f ./helm-overrides/gke-central-prd-ase1a/ai-gateway-ext/custom-values.yaml -n prd-ai-gateway-ext - -# -- Deployment Configuration -- -replicaCount: 2 - -fullnameOverride: "prd-ai-gateway-ext" - -image: - repository: docker.io/maximhq/bifrost - pullPolicy: IfNotPresent - tag: "v1.6.3" - -# -- Service Account -- -serviceAccount: - create: true - automount: true - annotations: - iam.gke.io/gcp-service-account: sa-dvops-ai-gateway@meesho-central-prd-0622.iam.gserviceaccount.com - name: "prd-ai-gateway-ext" - -# -- Pod Metadata -- -deploymentLabels: - bu: central - env: prod - team: devops - priority: p1 - priority_v2: sp1 - primary_owner: deep.shah - secondary_owner: anupam.satsangi - service: ai-gateway-ext - service_type: producer-httpstateless - -podLabels: - bu: central - env: prod - team: devops - priority: p1 - priority_v2: sp1 - primary_owner: deep.shah - secondary_owner: anupam.satsangi - service: ai-gateway-ext - service_type: producer-httpstateless - -podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: "8080" - prometheus.io/scrape: "true" - telegraf.influxdata.com/class: infra - -# -- Security Context -- -podSecurityContext: - fsGroup: 65534 - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - -securityContext: - capabilities: - drop: - - ALL - readOnlyRootFilesystem: false - runAsNonRoot: true - runAsUser: 65534 - -# -- Service -- -service: - type: ClusterIP - port: 8080 - -# -- Contour HTTPProxy -- -# ingress.enabled=false disables Bifrost's official K8s Ingress -# httpProxy.enabled=true enables the Meesho Contour HTTPProxy templates -# HTTPProxy templates read from ingress.* for hosts, class, etc. -httpProxy: - enabled: true -createContourGateway: true -namespace: prd-ai-gateway-ext -contourResponseTimeout: false -ingress: - enabled: false - ingressClassName: contour-external - servicePortNumber: 8080 - enableWebsocket: false - hosts: - - host: ai-gateway-ext.meeshogcp.in - paths: - - path: / - pathType: ImplementationSpecific - slowStart: - enabled: false - aggression: 1 - minPercent: 10 - window: 120s - -# -- Resources -- -resources: - limits: - cpu: "5" - memory: 10Gi - requests: - cpu: "4" - memory: 8Gi - -# -- Health Probes -- -livenessProbe: - httpGet: - path: /health - port: http - scheme: HTTP - initialDelaySeconds: 15 - periodSeconds: 5 - timeoutSeconds: 3 - failureThreshold: 5 - successThreshold: 1 - -readinessProbe: - httpGet: - path: /health - port: http - scheme: HTTP - initialDelaySeconds: 15 - periodSeconds: 5 - timeoutSeconds: 3 - failureThreshold: 5 - successThreshold: 1 - -# -- HPA (disabled - using KEDA) -- -autoscaling: - enabled: false - -# -- Scheduling -- -nodeSelector: - cloud.google.com/compute-class: megatetralite - -tolerations: - - key: cloud.google.com/compute-class - operator: Equal - value: megatetralite - effect: NoSchedule - -affinity: {} - -# -- Lifecycle & Graceful Shutdown -- -terminationGracePeriodSeconds: 300 -lifecycle: - preStop: - exec: - command: - - /bin/bash - - "-c" - - "kill -SIGQUIT; /bin/sleep 120" - -# -- Bifrost Application Config -- -bifrost: - appDir: /app/data - port: 8080 - host: 0.0.0.0 - logLevel: warn - logStyle: json - - # Auth configured via Bifrost UI (stored in DB), not in Helm values - # This avoids blocking /metrics scrape while still protecting the dashboard - - client: - dropExcessRequests: false - initialPoolSize: 300 - allowedOrigins: - - "*" - enableLogging: true - disableContentLogging: false - disableDbPingsInHealth: false - logRetentionDays: 365 - enforceGovernanceHeader: false - allowDirectKeys: false - maxRequestBodySizeMb: 100 - - # Configure providers with env.VAR_NAME references for API keys - # providers: - # openai: - # - keys: - # - value: "env.OPENAI_API_KEY" - # models: ["gpt-4o", "gpt-4o-mini"] - # weight: 1.0 - -# -- Storage (External PostgreSQL) -- -storage: - mode: postgres - configStore: - enabled: true - logsStore: - enabled: true - -postgresql: - enabled: false - external: - enabled: true - host: "env.BIFROST_POSTGRES_HOST" - port: 5432 - user: "env.BIFROST_POSTGRES_USER" - database: "env.BIFROST_POSTGRES_DATABASE" - sslMode: "disable" - existingSecret: "prd-ai-gateway-ext-vault" - passwordKey: "BIFROST_POSTGRES_PASSWORD" - -# -- Vector Store (disabled) -- -vectorStore: - enabled: false - type: none - -# -- Meesho Standard Env Vars -- -env: - - name: TZ - value: "Asia/Kolkata" - - name: BIFROST_POSTGRES_HOST - valueFrom: - secretKeyRef: - name: prd-ai-gateway-ext-vault - key: BIFROST_POSTGRES_HOST - - name: BIFROST_POSTGRES_USER - valueFrom: - secretKeyRef: - name: prd-ai-gateway-ext-vault - key: BIFROST_POSTGRES_USER - - name: BIFROST_POSTGRES_DATABASE - valueFrom: - secretKeyRef: - name: prd-ai-gateway-ext-vault - key: BIFROST_POSTGRES_DATABASE - - name: TELEGRAF_UDP_HOST - valueFrom: - fieldRef: - fieldPath: status.podIP - - name: NODE_IP - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - name: POD_NAME - valueFrom: - fieldRef: - fieldPath: metadata.name - - name: POD_NAMESPACE - valueFrom: - fieldRef: - fieldPath: metadata.namespace - - name: POD_IP - valueFrom: - fieldRef: - fieldPath: status.podIP - -# --- Meesho Infrastructure Extensions --- - -# -- PodDisruptionBudget -- -podDisruptionBudget: - enabled: true - maxUnavailable: "10%" - -# -- ExternalSecret (Vault) -- -# Creates K8s Secret "prd-ai-gateway-ext-vault" from Vault path -# This secret is referenced by postgresql.external.existingSecret above -externalSecret: - enabled: true - secretName: "prd-ai-gateway-ext-vault" - path: "prd/cntr/devop/ai-gateway-ext" - refreshInterval: "0" - secretStoreRef: "vault-backend" - -# -- KEDA ScaledObject -- -keda: - enabled: true - pollingInterval: 30 - minReplicaCount: 2 - maxReplicaCount: 200 - scaledown: - stabilizationWindowSeconds: 1800 - selectpolicy: Min - policies: - - type: Pods - value: 2 - periodseconds: 15 - scaleup: - stabilizationWindowSeconds: 120 - selectpolicy: Max - policies: - - type: Pods - value: 2 - periodseconds: 15 - - type: Percent - value: 10 - periodseconds: 15 - triggers: - - type: cpu - metricType: Utilization - metadata: - value: "40" diff --git a/helm-overrides/gke-central-prd-ase1a/ai-gateway/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/ai-gateway/custom-values.yaml deleted file mode 100644 index 7fcc37b..0000000 --- a/helm-overrides/gke-central-prd-ase1a/ai-gateway/custom-values.yaml +++ /dev/null @@ -1,286 +0,0 @@ -# Custom values for Bifrost (ai-gateway) - Meesho Production -# Usage: helm install bifrost ./helm-templates/bifrost/ -f ./helm-templates/bifrost/custom-values.yaml -n prd-ai-gateway - -# -- Deployment Configuration -- -replicaCount: 2 - -fullnameOverride: "prd-ai-gateway" - -image: - repository: docker.io/maximhq/bifrost - pullPolicy: IfNotPresent - tag: "v1.4.22" - -# -- Service Account -- -serviceAccount: - create: true - automount: true - annotations: - iam.gke.io/gcp-service-account: sa-dvops-ai-gateway@meesho-central-prd-0622.iam.gserviceaccount.com - name: "prd-ai-gateway" - -# -- Pod Metadata -- -deploymentLabels: - bu: central - env: prod - team: devops - priority: p1 - priority_v2: sp1 - primary_owner: deep.shah - secondary_owner: anupam.satsangi - service: ai-gateway - service_type: producer-httpstateless - -podLabels: - bu: central - env: prod - team: devops - priority: p1 - priority_v2: sp1 - primary_owner: deep.shah - secondary_owner: anupam.satsangi - service: ai-gateway - service_type: producer-httpstateless - -podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: "8080" - prometheus.io/scrape: "true" - telegraf.influxdata.com/class: infra - -# -- Security Context -- -podSecurityContext: - fsGroup: 65534 - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - -securityContext: - capabilities: - drop: - - ALL - readOnlyRootFilesystem: false - runAsNonRoot: true - runAsUser: 65534 - -# -- Service -- -service: - type: ClusterIP - port: 8080 - -# -- Contour HTTPProxy -- -# ingress.enabled=false disables Bifrost's official K8s Ingress -# httpProxy.enabled=true enables the Meesho Contour HTTPProxy templates -# HTTPProxy templates read from ingress.* for hosts, class, etc. -httpProxy: - enabled: true -createContourGateway: true -namespace: prd-ai-gateway -contourResponseTimeout: false -ingress: - enabled: false - ingressClassName: contour-internal-1 - servicePortNumber: 8080 - enableWebsocket: false - hosts: - - host: ai-gateway.prd.meesho.int - paths: - - path: / - pathType: ImplementationSpecific - - host: llm-gateway.prd.meesho.int - name: prd-llm-gateway-0 - intraName: prd-llm-gateway-intra-0 - paths: - - path: / - pathType: ImplementationSpecific - slowStart: - enabled: false - aggression: 1 - minPercent: 10 - window: 120s - -# -- Resources -- -resources: - limits: - cpu: "5" - memory: 25Gi - requests: - cpu: "4" - memory: 20Gi - -# -- Health Probes -- -livenessProbe: - httpGet: - path: /health - port: http - scheme: HTTP - initialDelaySeconds: 15 - periodSeconds: 5 - timeoutSeconds: 3 - failureThreshold: 5 - successThreshold: 1 - -readinessProbe: - httpGet: - path: /health - port: http - scheme: HTTP - initialDelaySeconds: 15 - periodSeconds: 5 - timeoutSeconds: 3 - failureThreshold: 5 - successThreshold: 1 - -# -- HPA (disabled - using KEDA) -- -autoscaling: - enabled: false - -# -- Scheduling -- -nodeSelector: - cloud.google.com/compute-class: megatetralite - -tolerations: - - key: cloud.google.com/compute-class - operator: Equal - value: megatetralite - effect: NoSchedule - -affinity: {} - -# -- Lifecycle & Graceful Shutdown -- -terminationGracePeriodSeconds: 300 -lifecycle: - preStop: - exec: - command: - - /bin/bash - - "-c" - - "kill -SIGQUIT; /bin/sleep 120" - -# -- Bifrost Application Config -- -bifrost: - appDir: /app/data - port: 8080 - host: 0.0.0.0 - logLevel: warn - logStyle: json - - # Auth configured via Bifrost UI (stored in DB), not in Helm values - # This avoids blocking /metrics scrape while still protecting the dashboard - - client: - dropExcessRequests: false - initialPoolSize: 300 - allowedOrigins: - - "*" - enableLogging: true - disableContentLogging: false - disableDbPingsInHealth: false - logRetentionDays: 365 - enforceGovernanceHeader: false - allowDirectKeys: false - maxRequestBodySizeMb: 100 - - # Configure providers with env.VAR_NAME references for API keys - # providers: - # openai: - # - keys: - # - value: "env.OPENAI_API_KEY" - # models: ["gpt-4o", "gpt-4o-mini"] - # weight: 1.0 - -# -- Storage (External PostgreSQL) -- -storage: - mode: postgres - configStore: - enabled: true - logsStore: - enabled: true - -postgresql: - enabled: false - external: - enabled: true - host: "10.147.2.236" - port: 5432 - user: "app_user_bifrost" - database: "bifrost_db" - sslMode: "disable" - existingSecret: "prd-ai-gateway-vault" - passwordKey: "BIFROST_POSTGRES_PASSWORD" - -# -- Vector Store (disabled) -- -vectorStore: - enabled: false - type: none - -# -- Meesho Standard Env Vars -- -env: - - name: TZ - value: "Asia/Kolkata" - - name: TELEGRAF_UDP_HOST - valueFrom: - fieldRef: - fieldPath: status.podIP - - name: NODE_IP - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - name: POD_NAME - valueFrom: - fieldRef: - fieldPath: metadata.name - - name: POD_NAMESPACE - valueFrom: - fieldRef: - fieldPath: metadata.namespace - - name: POD_IP - valueFrom: - fieldRef: - fieldPath: status.podIP - -# --- Meesho Infrastructure Extensions --- - -# -- PodDisruptionBudget -- -podDisruptionBudget: - enabled: true - maxUnavailable: "10%" - -# -- ExternalSecret (Vault) -- -# Creates K8s Secret "prd-ai-gateway-vault" from Vault path -# This secret is referenced by postgresql.external.existingSecret above -externalSecret: - enabled: true - secretName: "prd-ai-gateway-vault" - path: "prd/cntr/devop/ai-gateway" - refreshInterval: "0" - secretStoreRef: "vault-backend" - -# -- KEDA ScaledObject -- -keda: - enabled: true - pollingInterval: 30 - minReplicaCount: 2 - maxReplicaCount: 200 - scaledown: - stabilizationWindowSeconds: 1800 - selectpolicy: Min - policies: - - type: Pods - value: 2 - periodseconds: 15 - scaleup: - stabilizationWindowSeconds: 120 - selectpolicy: Max - policies: - - type: Pods - value: 2 - periodseconds: 15 - - type: Percent - value: 10 - periodseconds: 15 - triggers: - - type: cpu - metricType: Utilization - metadata: - value: "40" diff --git a/helm-overrides/gke-central-prd-ase1a/akamai-observability-mcp/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/akamai-observability-mcp/custom-values.yaml deleted file mode 100644 index e01ec87..0000000 --- a/helm-overrides/gke-central-prd-ase1a/akamai-observability-mcp/custom-values.yaml +++ /dev/null @@ -1,99 +0,0 @@ -# Akamai Observability MCP — grafana-mcp chart on gke-central-prd-ase1a -# Chart: helm-templates/grafana-mcp (v2.0.0+) - -fullnameOverride: "akamai-observability-mcp" - -replicas: 1 - -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/prd/devop/grafana-mcp - tag: "v2026-03-10" - pullPolicy: IfNotPresent - -labels: - bu: central - team: devops - service: akamai-observability-mcp - env: prd - -# -- Grafana connection. -# url: set this to the Akamai/observability Grafana endpoint reachable from -# gke-central-prd-ase1a (in-cluster DNS preferred; otherwise the prd FQDN). -# apiKeySecret: read GRAFANA_SERVICE_ACCOUNT_TOKEN from the K8s Secret produced -# by the ExternalSecret below. -grafana: - url: "" # TODO: set the Grafana base URL (e.g. https://grafana-akamai.prd.meesho.int) - apiKeySecret: - name: "akamai-observability-mcp-vault" - key: "GRAFANA_SERVICE_ACCOUNT_TOKEN" - -# -- Vault-backed secret. Mint the Grafana service-account token in the Grafana -# UI, store it at the path below under key GRAFANA_SERVICE_ACCOUNT_TOKEN, then -# ESO syncs it into the K8s Secret referenced above. -externalSecret: - enabled: true - secretName: "akamai-observability-mcp-vault" - path: "prd/cntr/devop/akamai-observability-mcp" # TODO: confirm Vault path - refreshInterval: "0" - secretStoreRef: "vault-backend" - -serviceAccount: - create: true - annotations: {} - -# SSE transport is long-lived — keep Contour from cutting connections. -contourResponseTimeout: "1h" - -createContourGateway: true - -ingress: - enabled: true - ingressClassName: contour-internal-1 - servicePortNumber: 8000 - enableWebsocket: true - hosts: - - host: akamai-observability-mcp.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 - -resources: - requests: - cpu: 250m - memory: 256Mi - limits: - cpu: 500m - memory: 512Mi - -# Scheduling — dedicated MCP node pool on gke-central-prd-ase1a. -nodeSelector: - cloud.google.com/compute-class: "devops-mcp" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "devops-mcp" - effect: NoSchedule - -securityContext: - fsGroup: 1000 - runAsNonRoot: true - runAsUser: 1000 - runAsGroup: 1000 - -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - readOnlyRootFilesystem: true - runAsNonRoot: true - runAsUser: 1000 - runAsGroup: 1000 diff --git a/helm-overrides/gke-central-prd-ase1a/aurva-dataplane/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/aurva-dataplane/custom-values.yaml deleted file mode 100644 index fc5add0..0000000 --- a/helm-overrides/gke-central-prd-ase1a/aurva-dataplane/custom-values.yaml +++ /dev/null @@ -1,689 +0,0 @@ -## Aurva Data Plane -## Ref: https://github.com/aurva-io/aurva-charts.git - -postgresql: - enabled: true - fullnameOverride: "aurva-dataplane-database" - volumePermissions: - ## @param volumePermissions.enabled Enable init container that changes the owner and group of the persistent volume - ## - enabled: true - global: - storageClass: hyperdisk-balanced - postgresql: - auth: - postgresPassword: "aurva" - database: "controller" - # Add toleration to make sure where this postgres db pod should reside (Applicable for production workloads): For more detail ref: https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/ - primary: - extendedConfiguration: | - max_connections = 300 - - #PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: central-devops - - # -- Select nodes to deploy which matches the following labels - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: central-devops - -# -- Provide a name in place of `aurva` -# namespaceOverride: aurva-dataplane - -########################################################## -# Global Configs -########################################################## -global: - aurva_controller: - enabled: true - aurva_fastdet: - enabled: false - aurva_pii_analyzer: - enabled: true - aurva_ocr: - enabled: true - aurva_collector: - enabled: true - - deploymentAnnotations: {} - - priorityClassName: "" - -########################################################## -# Aurva Controller -########################################################## -aurva_controller: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "central" - team: "central-devops" - service: "aurva-central-a-prd" - env: "prd" - priority: "p0" - type: "aurva_controller" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 10 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - image: - # Image of the app container - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-controller/aurva-controller - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - - type: secret - name: aurva-controller-secrets - # - type: configmap - # name: proxy-datasource-config - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 2 - requests: - memory: 1Gi - cpu: 1 - - aurvaFastdet: - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-fastdet/aurva-fastdet - tag: "v2.30.13" - pullPolicy: IfNotPresent - envFrom: [] - env: [] - resources: - limits: - cpu: 0.5 - memory: 512Mi - requests: - cpu: 0.5 - memory: 512Mi - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: central-devops - - # -- Taint tolerations for nodes - ##PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: central-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-controller-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - config: - #variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - COMMAND_URL: "command.aurva-prd.meeshogcp.in:80" - DEPLOYMENT_TYPE: "kubernetes" - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - FLUSHER_WORKER_POOL_SIZE: "1000" - FLUSHER_BATCH_SIZE: "30000" - UNIQUENESS_IDENTIFIER: "gke-central-prd-ase1a" #Recommendation: should be equal to cluster name - PROVIDER_ACCOUNT_ID: "meesho-central-prd-0622" # GCP PROJECT ID (not Number) - REGION: "asia-southeast1" #eg: asia-south1 - ENVIRONMENT: "prod" - OCR_ENABLED: "true" - MONITORING_ENABLED: "false" - HYBRID_ONLY_MODE: "false" - FORCE_TLS: "false" - FASTDET_FLAG : "true" - AADHAAR_ENHANCER: "0" - WORKSPACE_EVENT_TRACKING_ENABLED: "false" - ACCESS_IQ_ENABLED: "true" - GRPC_ENFORCE_ALPN_ENABLED: "false" - ENABLE_FASTDET: "true" - FASTDET_MAX_BATCH_SIZE: "100" - HEARTBEAT_INTERVAL: "5m" - #pii data - PII_BUCKET_NAME: "gcs-infra-devop-aurva-central-prd" - PII_LOG_BUCKET_REGION: "asia-southeast1" - PII_LOG_CRON: "*/5 * * * *" - ENABLE_PII_LOG: "true" - PII_EVIDENCE_UPLOAD_MAX_BLOCKING_TASKS: "50000" - PII_EVIDENCE_UPLOAD_MAX_CONCURRENT_TASKS: "500" - PII_EVIDENCE_MAX_CACHE_WEIGHT: "100" - QUOTA_CLEANUP_CRON: "0 0 * * *" - ENABLE_PII_QUOTA: "true" - MAX_PII_EVIDENCES_PER_KEY_PER_WINDOW: "3" - PII_QUOTA_SYNC_CRON: "*/2 * * * *" - QUOTA_WINDOW_HOURS: "24" - #constants - SKIP_NAMESPACES: "contour-internal-1-central-prd,contour-internal-0-central-prd,contour-external-central-prd" - CLOUD_PROVIDER: "gcp" - LOG_ENV: "production" - RDS_SCANNER_AVAILABILITY : "false" - REDSHIFT_SCANNER_AVAILABILITY : "false" - S3_SCANNER_AVAILABILITY : "false" - DYNAMO_SCANNER_AVAILABILITY: "false" - DOCDB_SCANNER_AVAILABILITY: "false" - OPENSEARCH_SCANNER_AVAILABILITY: "false" - CLOUDSQL_SCANNER_AVAILABILITY: "true" - BIGQUERY_SCANNER_AVAILABILITY: "true" - AWS_SNAPSHOT_SCANNER_AVAILABILITY: "false" - CLOUDSTORAGE_SCANNER_AVAILABILITY: "true" - KEYSPACES_SCANNER_AVAILABILITY: "false" - ALLOYDB_SCANNER_AVAILABILITY: "true" - BIGTABLE_SCANNER_AVAILABILITY: "true" - GCP_BACKUP_AVAILABILITY: "true" - EGRESS_MODE_ONLY: "false" - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - SCAN_UUID_ENABLED: "true" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-controller-sa - # -- Annotations applied to created service account - annotations: - iam.gke.io/gcp-service-account: sa-central-prd-aurva-contr@meesho-central-prd-0622.iam.gserviceaccount.com -# eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 15 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 - - -########################################################## -# Aurva OCR -########################################################## -aurva_ocr: - # -- Additional labels for aurva-controller - additionalLabels: - bu: "central" - team: "central-devops" - service: "aurva-central-a-prd" - env: "prd" - priority: "p0" - type: "aurva_ocr" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 1 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-ocr/aurva-ocr - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-ocr: - type: secret - name: aurva-ocr-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 1 - requests: - memory: 2Gi - cpu: 1 - - # -- Select nodes to deploy which matches the following labels - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: central-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: central-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-ocr-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - OCR_TIME_LIMIT: "1" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "gke-central-prd-ase1a" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-ocr-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # iam.gke.io/gcp-service-account: service-account@gcp.iam.gserviceaccount.com - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva Collector -########################################################## -aurva_collector: - - # -- Additional labels for aurva-analyzer - additionalLabels: - bu: "central" - team: "central-devops" - service: "aurva-central-a-prd" - env: "prd" - priority: "p0" - type: "aurva_collector" - - - # -- Annotations on aurva-analyzer - annotations: {} -# "key": "value" - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-collector/aurva-collector - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-controller: - type: secret - name: aurva-collector-secrets - - resources: - limits: - cpu: 800m - memory: 800Mi - requests: - cpu: 200m - memory: 512Mi - - podSecurityContext: {} - - securityContext: - privileged: true - capabilities: - add: - # For kernel v5.8 and above we don't need CAP_SYS_ADMIN or CAP_SYS_RESOURCE - # we just need CAP_BPF and CAP_PERFMON. This has been tested on our EKS node - # which is on kernel v5.10.x - # When SSL Tracing is required we need CAP_SYS_ADMIN and CAP_SYS_PTRACE - # on top of the previous capabilities - # So finally these are the 4 possible combinations for capabilies - # 1. Newer Kernels without SSL - # - BPF - # - PERFMON - # 2. Newer Kernels with SSL - - SYS_ADMIN - - SYS_PTRACE - # 3. Older Kernels without SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # 4. Older Kernels with SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # - SYS_PTRACE - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - volumes: - - name: debugfs - mountPath: /sys/kernel/debug - hostPath: /sys/kernel/debug - - name: vmlinux - mountPath: /sys/kernel/btf/vmlinux - hostPath: /sys/kernel/btf/vmlinux - - name: procfs - mountPath: /host/proc - hostPath: /proc - - name: bpffs - mountPath: /sys/fs/bpf - hostPath: /sys/fs/bpf - - # -- Taint tolerations for nodes - tolerations: - # - effect: NoSchedule - # key: dedicated - # operator: Equal - # value: megatetra - - operator: Exists - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmstorage-n4d - - vmselect-mds - - vmagent-mds - - vminsert-mds - - contour-internal-0 - - contour-internal-1 - - contour-external - - contour-internal-intra-1 - - contour-internal-intra-0 - - contour-external-arm - - contour-external-cc - - contour-internal-0-arm - - contour-internal-0-cc - - contour-internal-1-arm - - contour-internal-1-cc - - contour-intra-0-arm - - contour-intra-0-cc - - contour-intra-1-arm - - contour-intra-1-cc - - contour-shared-arm - - contour-shared-cc - - alloy - - preprod-spot-16 - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-collector-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - # variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-central-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - TRACE_INTERNAL_SVC: "true" - TRACE_INTERNAL_SVC_HTTP: "true" - INTERNAL_SVC_SAMPLE_INTERVAL: "10m" - LOGS_TTL: "1h" - TRACE_HTTP2: "true" - TRACE_SSL: "false" - TRACE_PSQL: "false" - TRACE_SQLSERVER: "false" - TRACE_MYSQL: "false" - TRACE_EGRESS: "true" - TRACE_GO_TLS: "false" - TRACE_ML_SERVICES: "false" - # constants - LOG_ENV: production - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - MONITORING_ENABLED: "false" - ENABLE_INGRESS_INFORMER: "false" - ENABLE_SERVICE_INFORMER: "false" - ENABLE_ISTIO_INFORMER: "false" - EXCLUDED_PII_REGEX_TYPES: "ip_address,us_bank_number,us_driver_license,us_itin,us_passport,us_routing,us_mbi,ssn" - AGGREGATOR_MAX_CONNECTIONS: "1000" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-collector-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva PII Analyzer -########################################################## -aurva_pii_analyzer: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "central" - team: "central-devops" - service: "aurva-central-a-prd" - env: "prd" - priority: "p0" - type: "aurva_pii_analyzer" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 3 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - ##PLACEHOLDER## - nodeSelector: - cloud.google.com/compute-class: central-devops - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-piianalyzer/aurva-piianalyzer - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-pii-analyzer: - type: secret - name: aurva-pii-analyzer-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 4Gi - cpu: 4 - requests: - memory: 2Gi - cpu: 2 - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: central-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: central-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-pii-analyzer-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - SCHEDULER_TIME: "1" - SUPPORTED_REGION: "US" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "gke-central-prd-ase1a" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-pii-analyzer-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 3 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 diff --git a/helm-overrides/gke-central-prd-ase1a/cert-manager/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/cert-manager/custom-values.yaml deleted file mode 100644 index 0c0befa..0000000 --- a/helm-overrides/gke-central-prd-ase1a/cert-manager/custom-values.yaml +++ /dev/null @@ -1,129 +0,0 @@ -global: - logLevel: 2 - rbac: - create: true - priorityClassName: "high-priority" -installCRDs: false - -crds: - enabled: true - keep: true - -# Cert-manager Controller -replicaCount: 1 -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-controller - tag: v1.20.1 - pullPolicy: IfNotPresent - -nodeSelector: - cloud.google.com/compute-class: central-devops - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - -resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 500m - memory: 512Mi - -# Webhook Configuration -webhook: - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-webhook - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: central-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# CA Injector Configuration -cainjector: - enabled: true - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-cainjector - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: central-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# ACME Solver Configuration -acmesolver: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-acmesolver - tag: v1.20.1 - pullPolicy: IfNotPresent - -# Startup API Check -startupapicheck: - enabled: true - timeout: 1m - backoffLimit: 4 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-startupapicheck - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: central-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - -# Prometheus Monitoring -prometheus: - enabled: true - servicemonitor: - enabled: false - interval: 60s - scrapeTimeout: 30s - labels: - prometheus: cert-manager diff --git a/helm-overrides/gke-central-prd-ase1a/clickhouse/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/clickhouse/custom-values.yaml deleted file mode 100644 index 97d7629..0000000 --- a/helm-overrides/gke-central-prd-ase1a/clickhouse/custom-values.yaml +++ /dev/null @@ -1,251 +0,0 @@ -# logHouse (central-prd) - overrides only. -# Google SSO via oauth2-proxy; ClickHouse ingress disabled. -oauth2Proxy: - enabled: true - -# nginx audit proxy maps X-Forwarded-Email → X-ClickHouse-Setting-log_comment -# so system.query_log.log_comment shows the SSO email of who ran each query. -auditProxy: - enabled: true - replicas: 2 - image: "nginx:1.27-alpine" - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "loghouse" - effect: "NoSchedule" - -externalSecret: - enabled: true - path: meesho/prd/cntr/devop/loghouse - secretName: loghouse-oauth2-secret - annotations: {} - -# --- Bitnami ClickHouse subchart --- -clickhouse: - replicaCount: 3 - global: - security: - allowInsecureImages: true - - image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/prd/sis/clickhouse - tag: 25.6.2-debian-12-r0 - - auth: - username: default - password: "" - existingSecret: "loghouse-oauth2-secret" - existingSecretKey: "clickhouse-password" - - - # Enable sampling so Bitnami's 08-sampling.xml preserves query_log, - # text_log, metric_log etc. All queries are recorded in system.query_log. - sampling: - enabled: true - - usersdFiles: - grant_all.xml: | - - - - 1 - 1 - - - - log_queries.xml: | - - - - 1 - 0 - - - - - initContainers: - - name: copy-usersd-config - image: busybox:1.36 - command: - - /bin/sh - - -ec - - cp -R /src/. /dst/ - volumeMounts: - - name: usersd-configuration-configuration - mountPath: /src - readOnly: true - - name: clickhouse-users-d - mountPath: /dst - - persistence: - storageClass: "hyperdisk-balanced" - size: 100Gi - mountPath: /var/lib/clickhouse - - - extraEnvVars: - - name: CLICKHOUSE_USER - value: "default" - - name: CLICKHOUSE_PASSWORD - valueFrom: - secretKeyRef: - name: loghouse-oauth2-secret - key: clickhouse-password - extraVolumes: - - name: clickhouse-users-d - emptyDir: - sizeLimit: 100Mi - - name: clickhouse-logs - emptyDir: - sizeLimit: 500Mi - - name: fluentbit-config - configMap: - name: loghouse-fluentbit-config - extraVolumeMounts: - - name: clickhouse-users-d - mountPath: /etc/clickhouse-server/users.d - - name: clickhouse-logs - mountPath: /var/log/clickhouse-server - - sidecars: - - name: query-log-tailer - image: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/prd/sis/clickhouse:25.6.2-debian-12-r0" - command: - - /bin/sh - - -c - - | - while true; do - clickhouse-client --host 127.0.0.1 --port 9000 --user default --password "$CLICKHOUSE_PASSWORD" --query="SELECT event_time, user, query_id, query, client_hostname FROM system.query_log WHERE type = 'QueryFinish' AND event_time > now() - INTERVAL 10 SECOND FORMAT JSONEachRow" 2>/dev/null; - sleep 10; - done - env: - - name: CLICKHOUSE_PASSWORD - valueFrom: - secretKeyRef: - name: loghouse-oauth2-secret - key: clickhouse-password - volumeMounts: - - name: clickhouse-logs - mountPath: /var/log/clickhouse-server - readOnly: true - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - - name: fluentbit - image: fluent/fluent-bit:3.1 - resources: - requests: - cpu: 25m - memory: 50Mi - limits: - cpu: 100m - memory: 100Mi - volumeMounts: - - name: clickhouse-logs - mountPath: /var/log/clickhouse-server - readOnly: true - - name: fluentbit-config - mountPath: /fluent-bit/etc - readOnly: true - - defaultInitContainers: - volumePermissions: - enabled: false - image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/prd/sis/os-shell - tag: 12-debian-12-r47 - - resourcesPreset: "none" - # Chart maps these inversely: values.requests -> pod limits, values.limits -> pod requests - resources: - requests: - cpu: "6" - memory: 40Gi - limits: - cpu: "6" - memory: 40Gi - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "loghouse" - effect: "NoSchedule" - - # Disabled when oauth2Proxy.enabled is true (oauth2-proxy handles ingress) - ingress: - enabled: false - - networkPolicy: - enabled: true - allowExternal: true - allowExternalEgress: true - - keeper: - enabled: false - - -oauth2-proxy: - replicaCount: 3 - config: - existingSecret: loghouse-oauth2-secret - requiredSecretKeys: - - client-id - - client-secret - - cookie-secret - extraArgs: - provider: google - redirect-url: "http://loghouse-central.prd.meesho.int/oauth2/callback" - upstream: "http://loghouse-central-a-prd-audit-proxy:8123" - email-domain: "meesho.com" - proxy-prefix: "/oauth2" - pass-host-header: "true" - proxy-websockets: "true" - real-client-ip-header: "X-Forwarded-For" - cookie-secure: "false" - cookie-expire: "0s" - custom-templates-dir: "/templates" - skip-jwt-bearer-tokens: "true" - oidc-issuer-url: "https://accounts.google.com" - extra-jwt-issuers: "https://accounts.google.com=32555940559.apps.googleusercontent.com" - pass-user-headers: "true" - set-xauthrequest: "true" - request-logging: "true" - auth-logging: "true" - standard-logging: "true" - extraVolumes: - - name: custom-templates - configMap: - name: '{{ .Release.Name }}-oauth2-proxy-templates' - extraVolumeMounts: - - name: custom-templates - mountPath: /templates - readOnly: true - service: - portNumber: 80 - ingress: - enabled: true - className: contour-internal-1 - path: / - pathType: Prefix - hosts: - - loghouse-central.prd.meesho.int - annotations: {} - tls: [] - sessionStorage: - type: cookie - redis-ha: - enabled: false - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 200m - memory: 128Mi diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/azul-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/azul-cc.yaml deleted file mode 100644 index 7a3a3c3..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/azul-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: azul -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: azul - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c2-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: c2-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: c2-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/central-devops-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/central-devops-cc.yaml deleted file mode 100644 index e50a670..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/central-devops-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: central-devops -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: central-devops - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: e2-standard-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: e2-standard-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: e2-standard-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/central-kyverno-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/central-kyverno-cc.yaml deleted file mode 100644 index 8b596d5..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/central-kyverno-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: central-kyverno -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: central-kyverno - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/compactduo-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/compactduo-cc.yaml deleted file mode 100644 index ff087ff..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/compactduo-cc.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: compactduo -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: compactduo - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c4-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-a - machineType: c3-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: c4-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: c3-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: c4-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: c3-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/compacttetra-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/compacttetra-cc.yaml deleted file mode 100644 index 23d8066..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/compacttetra-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: compacttetra -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: compacttetra - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: c3-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: c3-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-external-arm.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/contour-external-arm.yaml deleted file mode 100644 index 4a4a31f..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-external-arm.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-external-arm -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-external-arm - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-a - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-external-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/contour-external-cc.yaml deleted file mode 100644 index 93a6cd9..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-external-cc.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-external-cc -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-external-cc - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-a - machineType: c4d-highcpu-8 - spot: false - maxPodsPerNode: 16 - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-a - machineType: n2d-highcpu-8 - spot: false - maxPodsPerNode: 16 - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: c4d-highcpu-8 - spot: false - maxPodsPerNode: 16 - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-0-arm.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-0-arm.yaml deleted file mode 100644 index f8e3af1..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-0-arm.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-0-arm -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-internal-0-arm - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-a - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-0-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-0-cc.yaml deleted file mode 100644 index 9b83228..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-0-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-0-cc -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-internal-0-cc - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-1-arm.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-1-arm.yaml deleted file mode 100644 index 7351ae7..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-1-arm.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-1-arm -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-internal-1-arm - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-a - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-1-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-1-cc.yaml deleted file mode 100644 index 7492634..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-1-cc.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-1-cc -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-internal-1-cc - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-a - machineType: c4d-highcpu-8 - spot: false - maxPodsPerNode: 16 - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-a - machineType: n2d-highcpu-8 - spot: false - maxPodsPerNode: 16 - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: c4d-highcpu-8 - spot: false - maxPodsPerNode: 16 - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-intra-0-arm.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-intra-0-arm.yaml deleted file mode 100644 index 041fae7..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-intra-0-arm.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-0-arm -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-intra-0-arm - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-a - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-intra-0-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-intra-0-cc.yaml deleted file mode 100644 index b828df0..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-intra-0-cc.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-0-cc -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-intra-0-cc - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-a - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-intra-1-arm.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-intra-1-arm.yaml deleted file mode 100644 index a6c6122..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-intra-1-arm.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-1-arm -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-intra-1-arm - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-a - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-intra-1-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-intra-1-cc.yaml deleted file mode 100644 index 63d86fc..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-internal-intra-1-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-1-cc -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-intra-1-cc - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-shared-arm.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/contour-shared-arm.yaml deleted file mode 100644 index 58422c1..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-shared-arm.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-shared-arm -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-shared-arm - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-a - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-shared-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/contour-shared-cc.yaml deleted file mode 100644 index a524a4d..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/contour-shared-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-shared-cc -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-shared-cc - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/devops-mcp-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/devops-mcp-cc.yaml deleted file mode 100644 index 75c8e09..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/devops-mcp-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: devops-mcp -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: devops-mcp - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2d-highcpu-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2d-highcpu-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2d-highcpu-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/gatekeeper-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/gatekeeper-cc.yaml deleted file mode 100644 index 7ea2573..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/gatekeeper-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: gatekeeper -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: gatekeeper - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3-highcpu-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: c3-highcpu-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: c3-highcpu-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/loghouse-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/loghouse-cc.yaml deleted file mode 100644 index 243913e..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/loghouse-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: loghouse -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: loghouse - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3d-highmem-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: c3d-highmem-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: c3d-highmem-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/megaduo-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/megaduo-cc.yaml deleted file mode 100644 index e575c88..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/megaduo-cc.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaduo -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megaduo - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3-highcpu-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-a - machineType: c4-highcpu-24 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: c3-highcpu-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: c4-highcpu-24 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: c3-highcpu-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: c4-highcpu-24 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/megaduolite-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/megaduolite-cc.yaml deleted file mode 100644 index d4e9d6f..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/megaduolite-cc.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaduolite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megaduolite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-a - machineType: n2-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n4-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n2-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n4-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n2-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/megaoctalite-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/megaoctalite-cc.yaml deleted file mode 100644 index 4dd4894..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/megaoctalite-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaoctalite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megaoctalite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-highmem-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2-highmem-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2-highmem-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/megatetra-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/megatetra-cc.yaml deleted file mode 100644 index a0d65be..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/megatetra-cc.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megatetra -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megatetra - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-a - machineType: c3-standard-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: c3-standard-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: c3-standard-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/megatetralite-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/megatetralite-cc.yaml deleted file mode 100644 index 030fec6..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/megatetralite-cc.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megatetralite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megatetralite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-a - machineType: n2-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n4-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n2-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n4-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n2-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/megauno-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/megauno-cc.yaml deleted file mode 100644 index 0fbc0c6..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/megauno-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megauno -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megauno - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/megaunolite-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/megaunolite-cc.yaml deleted file mode 100644 index 657207c..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/megaunolite-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaunolite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megaunolite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-highcpu-48 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2-highcpu-48 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2-highcpu-48 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/mlp-g2-standard-8-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/mlp-g2-standard-8-cc.yaml deleted file mode 100644 index b5eb83a..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/mlp-g2-standard-8-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: mlp-g2-standard-8 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: mlp-g2-standard-8 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: g2-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: g2-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: g2-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/sale-rescue-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/sale-rescue-cc.yaml deleted file mode 100644 index ca1750f..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/sale-rescue-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sale-rescue -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sale-rescue - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3-standard-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: c3-standard-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: c3-standard-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/session-mgr-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/session-mgr-cc.yaml deleted file mode 100644 index 53d07b8..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/session-mgr-cc.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: session-mgr -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: session-mgr - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3d-highcpu-60 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-a - machineType: c3d-standard-60 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: c3d-highcpu-60 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: c3d-standard-60 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: c3d-highcpu-60 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: c3d-standard-60 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/sumoduo-c4d-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/sumoduo-c4d-cc.yaml deleted file mode 100644 index 33e5748..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/sumoduo-c4d-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumoduo-c4d -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumoduo-c4d - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c4d-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: c4d-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: c4d-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/sumoduo-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/sumoduo-cc.yaml deleted file mode 100644 index 8dc3707..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/sumoduo-cc.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumoduo -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumoduo - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2d-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-a - machineType: n2d-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-a - machineType: n4d-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n2d-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2d-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2d-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/sumoduolite-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/sumoduolite-cc.yaml deleted file mode 100644 index ab9538e..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/sumoduolite-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumoduolite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumoduolite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-custom-32-65536 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2-custom-32-65536 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2-custom-32-65536 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/sumotetra-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/sumotetra-cc.yaml deleted file mode 100644 index cdfa187..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/sumotetra-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumotetra -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumotetra - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/sumouno-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/sumouno-cc.yaml deleted file mode 100644 index f0e3596..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/sumouno-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumouno -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumouno - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-highcpu-48 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2-highcpu-48 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2-highcpu-48 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/sumounolite-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/sumounolite-cc.yaml deleted file mode 100644 index 7cf83ce..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/sumounolite-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumounolite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumounolite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-highcpu-48 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2-highcpu-48 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2-highcpu-48 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/vmagent-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/vmagent-cc.yaml deleted file mode 100644 index 3e2a202..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/vmagent-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmagent -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmagent - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/vmagent-dr-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/vmagent-dr-cc.yaml deleted file mode 100644 index 14c26c5..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/vmagent-dr-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmagent-dr -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmagent-dr - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/vmagent-mds-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/vmagent-mds-cc.yaml deleted file mode 100644 index 45c0fe2..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/vmagent-mds-cc.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmagent-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmagent-mds - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-a - machineType: n4-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/vminsert-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/vminsert-cc.yaml deleted file mode 100644 index d36a08d..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/vminsert-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vminsert -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vminsert - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/vminsert-mds-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/vminsert-mds-cc.yaml deleted file mode 100644 index 811fcb2..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/vminsert-mds-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vminsert-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vminsert-mds - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/vmselect-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/vmselect-cc.yaml deleted file mode 100644 index ce2e1b2..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/vmselect-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmselect -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmselect - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/vmselect-mds-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/vmselect-mds-cc.yaml deleted file mode 100644 index 118883b..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/vmselect-mds-cc.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmselect-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmselect-mds - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3-highcpu-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-a - machineType: c4-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: c3-highcpu-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: c4-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: c3-highcpu-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: c4-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-cc.yaml deleted file mode 100644 index 5983852..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmstorage - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-highmem-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2-highmem-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2-highmem-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-mds-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-mds-cc.yaml deleted file mode 100644 index a569c9e..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-mds-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmstorage-mds - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-highmem-48 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2-highmem-48 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2-highmem-48 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml deleted file mode 100644 index 447b4ff..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage-n4d -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmstorage-n4d - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4d-highmem-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4d-highmem-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4d-highmem-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-sale-24aug-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-sale-24aug-cc.yaml deleted file mode 100644 index 46df6d4..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-sale-24aug-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage-sale-24aug -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmstorage-sale-24aug - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-highmem-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2-highmem-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2-highmem-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-sale-cc.yaml b/helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-sale-cc.yaml deleted file mode 100644 index 58d5f33..0000000 --- a/helm-overrides/gke-central-prd-ase1a/computeclass/vmstorage-sale-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage-sale -spec: - nodePoolConfig: - serviceAccount: sa-common-np-cntr-prd@meesho-central-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmstorage-sale - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-highmem-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n2-highmem-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n2-highmem-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-central-prd-ase1a/conntrack-adjuster/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index ebd3809..0000000 --- a/helm-overrides/gke-central-prd-ase1a/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,25 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - max: 2097152 - hashsize: 524288 - sleepInterval: 30 - -nodeAffinity: - enabled: true - nodeSelectorTerms: - - matchExpressions: - - key: cloud.google.com/compute-class - operator: In - values: - - "contour-external-cc" - - "contour-internal-0-cc" - - "contour-internal-1-cc" - - "contour-intra-0-cc" - - "contour-intra-1-cc" - - matchExpressions: - - key: dedicated - operator: In - values: - - "contour-external-1" diff --git a/helm-overrides/gke-central-prd-ase1a/contour-ca-issuer/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/contour-ca-issuer/custom-values.yaml deleted file mode 100644 index 9212add..0000000 --- a/helm-overrides/gke-central-prd-ase1a/contour-ca-issuer/custom-values.yaml +++ /dev/null @@ -1,17 +0,0 @@ -# ClusterIssuer override for k8s-demand-prd-ase1 (prd demand cluster, ase1a zone). -# ClusterIssuer is a cluster-scoped resource — this file pins the issuer -# identity for this cluster so the name is auditable per-cluster. -# -# Must match the `issuerRef.name` in consuming Certificate CRs (see -# devops-helm-charts/2.0.0/templates/proxyless-grpc-cert.yaml). - -issuerName: contour-central-prd-ca-issuer -rootCASecretName: contour-central-ca - -externalSecret: - enabled: true - vaultPath: meesho/devops/contour/root-ca - refreshInterval: "0" - secretStoreRef: - name: vault-backend -namespace: cert-manager-central-prd diff --git a/helm-overrides/gke-central-prd-ase1a/contour-cert-checker/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/contour-cert-checker/custom-values.yaml deleted file mode 100644 index e3b0588..0000000 --- a/helm-overrides/gke-central-prd-ase1a/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "0 12 * * *" - args: ["--cluster=gke-central-prd-ase1a"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - cloud.google.com/compute-class: central-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: central-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/gke-central-prd-ase1a/contour-external-1/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/contour-external-1/custom-values.yaml deleted file mode 100644 index 67f1185..0000000 --- a/helm-overrides/gke-central-prd-ase1a/contour-external-1/custom-values.yaml +++ /dev/null @@ -1,101 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external-1 - nodeSelector: - dedicated: contour-external-1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external-1 - nodeSelector: - dedicated: contour-external-1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-1-central-ase1a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/gke-central-prd-ase1a/contour-external/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/contour-external/custom-values.yaml deleted file mode 100644 index 9c7e575..0000000 --- a/helm-overrides/gke-central-prd-ase1a/contour-external/custom-values.yaml +++ /dev/null @@ -1,106 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-external-arm - nodeSelector: - cloud.google.com/compute-class: contour-external-arm - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 6' - resources: - requests: - cpu: 6 - memory: 3Gi - limits: - cpu: 6 - memory: 12Gi - service: - export: - enabled: false # disabled for gke-central-prd-ase1a onboarding — re-enable before traffic shift - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-central-ase1a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/gke-central-prd-ase1a/contour-internal-0/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/contour-internal-0/custom-values.yaml deleted file mode 100644 index 5c466cc..0000000 --- a/helm-overrides/gke-central-prd-ase1a/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,109 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-0-arm - nodeSelector: - cloud.google.com/compute-class: contour-internal-0-arm - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 6' - resources: - requests: - cpu: 6 - memory: 3Gi - limits: - cpu: 6 - memory: 12Gi - service: - tcpLB: true # disabled for gke-central-prd-ase1a onboarding — re-enable before traffic shift - export: - enabled: true # disabled for gke-central-prd-ase1a onboarding — re-enable before traffic shift - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-central-ase1a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/gke-central-prd-ase1a/contour-internal-1/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/contour-internal-1/custom-values.yaml deleted file mode 100644 index d4f03ac..0000000 --- a/helm-overrides/gke-central-prd-ase1a/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,113 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 360s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-1-arm - - effect: NoSchedule - key: kubernetes.io/arch - operator: Equal - value: arm64 - nodeSelector: - cloud.google.com/compute-class: contour-internal-1-arm - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 6' - resources: - requests: - cpu: 6 - memory: 3Gi - limits: - cpu: 6 - memory: 12Gi - service: - tcpLB: true # disabled for gke-central-prd-ase1a onboarding — re-enable before traffic shift - export: - enabled: true # disabled for gke-central-prd-ase1a onboarding — re-enable before traffic shift - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int1-central-ase1a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/gke-central-prd-ase1a/contour-internal-intra-0/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index 8d3e9a5..0000000 --- a/helm-overrides/gke-central-prd-ase1a/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,106 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-intra-0-arm - nodeSelector: - cloud.google.com/compute-class: contour-intra-0-arm - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 6' - resources: - requests: - cpu: 6 - memory: 3Gi - limits: - cpu: 6 - memory: 12Gi - service: - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/gke-central-prd-ase1a/contour-internal-intra-1/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index 0591f91..0000000 --- a/helm-overrides/gke-central-prd-ase1a/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,107 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 360s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-intra-1-arm - nodeSelector: - cloud.google.com/compute-class: contour-intra-1-arm - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 6' - resources: - requests: - cpu: 6 - memory: 3Gi - limits: - cpu: 6 - memory: 12Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/gke-central-prd-ase1a/coredns/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/coredns/custom-values.yaml deleted file mode 100644 index 5c03a9d..0000000 --- a/helm-overrides/gke-central-prd-ase1a/coredns/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -replicaCount: 24 -communicationType: "intra" - -labels: - bu: central - team: central-devops - env: prd - -clusterIP: 10.1.80.2 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - -nodeSelector: - cloud.google.com/compute-class: central-devops - kubernetes.io/os: linux diff --git a/helm-overrides/gke-central-prd-ase1a/elasticsearch-mcp/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/elasticsearch-mcp/custom-values.yaml deleted file mode 100644 index 8bc758a..0000000 --- a/helm-overrides/gke-central-prd-ase1a/elasticsearch-mcp/custom-values.yaml +++ /dev/null @@ -1,95 +0,0 @@ -fullnameOverride: "elasticsearch-mcp" - -replicas: 1 - -image: - repository: docker.elastic.co/mcp/elasticsearch - tag: "0.4.6" - pullPolicy: IfNotPresent - -labels: - bu: central - team: devops - service: elasticsearch-mcp - env: prd - -serviceAccount: - create: true - annotations: {} - -podAnnotations: {} - -podSecurityContext: {} - -securityContext: - runAsNonRoot: true - runAsUser: 1000 - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -priorityClassName: "" - -nodeSelector: - cloud.google.com/compute-class: "devops-mcp" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "devops-mcp" - effect: NoSchedule - -affinity: {} - -resources: - requests: - cpu: 500m - memory: 1Gi - limits: - cpu: 2000m - memory: 2Gi - -livenessProbe: - httpGet: - path: / - port: 8080 - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - -readinessProbe: - httpGet: - path: / - port: 8080 - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - -externalSecrets: - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/prd/cntr/devop/elasticsearch-mcp" - -createContourGateway: true - -ingress: - enabled: true - ingressClassName: contour-internal-1 - servicePortNumber: 8080 - hosts: - - host: elastic-mcp.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 diff --git a/helm-overrides/gke-central-prd-ase1a/etcd/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/etcd/custom-values.yaml deleted file mode 100644 index de366c3..0000000 --- a/helm-overrides/gke-central-prd-ase1a/etcd/custom-values.yaml +++ /dev/null @@ -1,1105 +0,0 @@ -# Copyright Broadcom, Inc. All Rights Reserved. -# SPDX-License-Identifier: APACHE-2.0 - -## @section Global parameters -## Global Docker image parameters -## Please, note that this will override the image parameters, including dependencies, configured to use the global value -## Current available global Docker image parameters: imageRegistry, imagePullSecrets and storageClass -## - -## @param global.imageRegistry Global Docker image registry -## @param global.imagePullSecrets [array] Global Docker registry secret names as an array -## @param global.defaultStorageClass Global default StorageClass for Persistent Volume(s) -## @param global.storageClass DEPRECATED: use global.defaultStorageClass instead -## -global: - imageRegistry: "" - ## E.g. - ## imagePullSecrets: - ## - myRegistryKeySecretName - ## - imagePullSecrets: [] - defaultStorageClass: "" - storageClass: "" - ## Compatibility adaptations for Kubernetes platforms - ## - compatibility: - ## Compatibility adaptations for Openshift - ## - openshift: - ## @param global.compatibility.openshift.adaptSecurityContext Adapt the securityContext sections of the deployment to make them compatible with Openshift restricted-v2 SCC: remove runAsUser, runAsGroup and fsGroup and let the platform use their allowed default IDs. Possible values: auto (apply if the detected running cluster is Openshift), force (perform the adaptation always), disabled (do not perform adaptation) - ## - adaptSecurityContext: auto -## @section Common parameters -## - -## @param kubeVersion Force target Kubernetes version (using Helm capabilities if not set) -## -kubeVersion: "" -## @param nameOverride String to partially override common.names.fullname template (will maintain the release name) -## -nameOverride: "" -## @param fullnameOverride String to fully override common.names.fullname template -## -fullnameOverride: "" -## @param commonLabels [object] Labels to add to all deployed objects -## -commonLabels: {} -## @param commonAnnotations [object] Annotations to add to all deployed objects -## -commonAnnotations: {} -## @param clusterDomain Default Kubernetes cluster domain -## -clusterDomain: cluster.local -## @param extraDeploy [array] Array of extra objects to deploy with the release -## -extraDeploy: [] -## Enable diagnostic mode in the deployment -## -diagnosticMode: - ## @param diagnosticMode.enabled Enable diagnostic mode (all probes will be disabled and the command will be overridden) - ## - enabled: false - ## @param diagnosticMode.command Command to override all containers in the deployment - ## - command: - - sleep - ## @param diagnosticMode.args Args to override all containers in the deployment - ## - args: - - infinity -## @section etcd parameters -## - -## Bitnami etcd image version -## ref: https://hub.docker.com/r/bitnami/etcd/tags/ -## @param image.registry [default: REGISTRY_NAME] etcd image registry -## @param image.repository [default: REPOSITORY_NAME/etcd] etcd image name -## @skip image.tag etcd image tag -## @param image.digest etcd image digest in the way sha256:aa.... Please note this parameter, if set, will override the tag -## -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/devops/bitnami/etcd - tag: 3.5.16-debian-12-r2 - digest: "" - ## @param image.pullPolicy etcd image pull policy - ## Specify a imagePullPolicy - ## Defaults to 'Always' if image tag is 'latest', else set to 'IfNotPresent' - ## ref: https://kubernetes.io/docs/concepts/containers/images/#pre-pulled-images - ## - pullPolicy: IfNotPresent - ## @param image.pullSecrets [array] etcd image pull secrets - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## e.g: - ## pullSecrets: - ## - myRegistryKeySecretName - ## - pullSecrets: [] - ## @param image.debug Enable image debug mode - ## Set to true if you would like to see extra information on logs - ## - debug: false -## Authentication parameters -## -auth: - ## Role-based access control parameters - ## ref: https://etcd.io/docs/current/op-guide/authentication/ - ## - rbac: - ## @param auth.rbac.create Switch to enable RBAC authentication - ## - create: true - ## @param auth.rbac.allowNoneAuthentication Allow to use etcd without configuring RBAC authentication - ## - allowNoneAuthentication: true - ## @param auth.rbac.rootPassword Root user password. The root user is always `root` - ## - rootPassword: "" - ## @param auth.rbac.existingSecret Name of the existing secret containing credentials for the root user - ## - existingSecret: "etcd-root-password" - ## @param auth.rbac.existingSecretPasswordKey Name of key containing password to be retrieved from the existing secret - ## - existingSecretPasswordKey: "rootPassword" - ## Authentication token - ## ref: https://etcd.io/docs/latest/learning/design-auth-v3/#two-types-of-tokens-simple-and-jwt - ## - token: - ## @param auth.token.enabled Enables token authentication - ## - enabled: true - ## @param auth.token.type Authentication token type. Allowed values: 'simple' or 'jwt' - ## ref: https://etcd.io/docs/latest/op-guide/configuration/#--auth-token - ## - type: jwt - ## @param auth.token.privateKey.filename Name of the file containing the private key for signing the JWT token - ## @param auth.token.privateKey.existingSecret Name of the existing secret containing the private key for signing the JWT token - ## NOTE: Ignored if auth.token.type=simple - ## NOTE: A secret containing a private key will be auto-generated if an existing one is not provided. - ## - privateKey: - filename: jwt-token.pem - existingSecret: "" - ## @param auth.token.signMethod JWT token sign method - ## NOTE: Ignored if auth.token.type=simple - ## - signMethod: RS256 - ## @param auth.token.ttl JWT token TTL - ## NOTE: Ignored if auth.token.type=simple - ## - ttl: 10m - ## TLS authentication for client-to-server communications - ## ref: https://etcd.io/docs/current/op-guide/security/ - ## - client: - ## @param auth.client.secureTransport Switch to encrypt client-to-server communications using TLS certificates - ## - secureTransport: false - ## @param auth.client.useAutoTLS Switch to automatically create the TLS certificates - ## - useAutoTLS: false - ## @param auth.client.existingSecret Name of the existing secret containing the TLS certificates for client-to-server communications - ## - existingSecret: "" - ## @param auth.client.enableAuthentication Switch to enable host authentication using TLS certificates. Requires existing secret - ## - enableAuthentication: false - ## @param auth.client.certFilename Name of the file containing the client certificate - ## - certFilename: cert.pem - ## @param auth.client.certKeyFilename Name of the file containing the client certificate private key - ## - certKeyFilename: key.pem - ## @param auth.client.caFilename Name of the file containing the client CA certificate - ## If not specified and `auth.client.enableAuthentication=true` or `auth.rbac.enabled=true`, the default is is `ca.crt` - ## - caFilename: "" - ## TLS authentication for server-to-server communications - ## ref: https://etcd.io/docs/current/op-guide/security/ - ## - peer: - ## @param auth.peer.secureTransport Switch to encrypt server-to-server communications using TLS certificates - ## - secureTransport: false - ## @param auth.peer.useAutoTLS Switch to automatically create the TLS certificates - ## - useAutoTLS: false - ## @param auth.peer.existingSecret Name of the existing secret containing the TLS certificates for server-to-server communications - ## - existingSecret: "" - ## @param auth.peer.enableAuthentication Switch to enable host authentication using TLS certificates. Requires existing secret - ## - enableAuthentication: false - ## @param auth.peer.certFilename Name of the file containing the peer certificate - ## - certFilename: cert.pem - ## @param auth.peer.certKeyFilename Name of the file containing the peer certificate private key - ## - certKeyFilename: key.pem - ## @param auth.peer.caFilename Name of the file containing the peer CA certificate - ## If not specified and `auth.peer.enableAuthentication=true` or `rbac.enabled=true`, the default is is `ca.crt` - ## - caFilename: "" -## @param autoCompactionMode Auto compaction mode, by default periodic. Valid values: "periodic", "revision". -## - 'periodic' for duration based retention, defaulting to hours if no time unit is provided (e.g. 5m). -## - 'revision' for revision number based retention. -## -autoCompactionMode: "" -## @param autoCompactionRetention Auto compaction retention for mvcc key value store in hour, by default 0, means disabled -## -autoCompactionRetention: "" -## @param initialClusterState Initial cluster state. Allowed values: 'new' or 'existing' -## If this values is not set, the default values below are set: -## - 'new': when installing the chart ('helm install ...') -## - 'existing': when upgrading the chart ('helm upgrade ...') -## -initialClusterState: "" -## @param initialClusterToken Initial cluster token. Can be used to protect etcd from cross-cluster-interaction, which might corrupt the clusters. -## If spinning up multiple clusters (or creating and destroying a single cluster) -## with same configuration for testing purpose, it is highly recommended that each cluster is given a unique initial-cluster-token. -## By doing this, etcd can generate unique cluster IDs and member IDs for the clusters even if they otherwise have the exact same configuration. -## -initialClusterToken: "etcd-cluster-k8s" -## @param logLevel Sets the log level for the etcd process. Allowed values: 'debug', 'info', 'warn', 'error', 'panic', 'fatal' -## -logLevel: "info" -## @param maxProcs Limits the number of operating system threads that can execute user-level -## Go code simultaneously by setting GOMAXPROCS environment variable -## ref: https://golang.org/pkg/runtime -## -maxProcs: "" -## @param removeMemberOnContainerTermination Use a PreStop hook to remove the etcd members from the etcd cluster on container termination -## they the containers are terminated. Set to 'false' if appears an error-related member ID wasn't properly stored. -## NOTE: Ignored if lifecycleHooks is set or replicaCount=1 -## -removeMemberOnContainerTermination: true -## @param configuration etcd configuration. Specify content for etcd.conf.yml -## e.g: -## configuration: |- -## foo: bar -## baz: -## -configuration: "" -## @param existingConfigmap Existing ConfigMap with etcd configuration -## NOTE: When it's set the configuration parameter is ignored -## -existingConfigmap: "" -## @param extraEnvVars [array] Extra environment variables to be set on etcd container -## e.g: -## extraEnvVars: -## - name: FOO -## value: "bar" -## -extraEnvVars: [] -## @param extraEnvVarsCM Name of existing ConfigMap containing extra env vars -## -extraEnvVarsCM: "" -## @param extraEnvVarsSecret Name of existing Secret containing extra env vars -## -extraEnvVarsSecret: "" -## @param command [array] Default container command (useful when using custom images) -## -command: [] -## @param args [array] Default container args (useful when using custom images) -## -args: [] -## @section etcd statefulset parameters -## - -## @param replicaCount Number of etcd replicas to deploy -## -replicaCount: 1 -## Update strategy -## ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies -## @param updateStrategy.type Update strategy type, can be set to RollingUpdate or OnDelete. -## -updateStrategy: - type: RollingUpdate -## @param podManagementPolicy Pod management policy for the etcd statefulset -## ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#pod-management-policies -## -podManagementPolicy: Parallel -## @param automountServiceAccountToken Mount Service Account token in pod -## -automountServiceAccountToken: false -## @param hostAliases [array] etcd pod host aliases -## ref: https://kubernetes.io/docs/concepts/services-networking/add-entries-to-pod-etc-hosts-with-host-aliases/ -## -hostAliases: [] -## @param lifecycleHooks [object] Override default etcd container hooks -## -lifecycleHooks: {} -## etcd container ports to open -## @param containerPorts.client Client port to expose at container level -## @param containerPorts.peer Peer port to expose at container level -## @param containerPorts.metrics Metrics port to expose at container level when metrics.useSeparateEndpoint is true -## -containerPorts: - client: 2379 - peer: 2380 - metrics: 9090 -## etcd pods' Security Context -## ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-pod -## @param podSecurityContext.enabled Enabled etcd pods' Security Context -## @param podSecurityContext.fsGroupChangePolicy Set filesystem group change policy -## @param podSecurityContext.sysctls Set kernel settings using the sysctl interface -## @param podSecurityContext.supplementalGroups Set filesystem extra groups -## @param podSecurityContext.fsGroup Set etcd pod's Security Context fsGroup -## -podSecurityContext: - enabled: true - fsGroupChangePolicy: Always - sysctls: [] - supplementalGroups: [] - fsGroup: 1001 -## etcd containers' SecurityContext -## ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -## @param containerSecurityContext.enabled Enabled etcd containers' Security Context -## @param containerSecurityContext.seLinuxOptions [object,nullable] Set SELinux options in container -## @param containerSecurityContext.runAsUser Set etcd containers' Security Context runAsUser -## @param containerSecurityContext.runAsGroup Set etcd containers' Security Context runAsUser -## @param containerSecurityContext.runAsNonRoot Set Controller container's Security Context runAsNonRoot -## @param containerSecurityContext.privileged Set primary container's Security Context privileged -## @param containerSecurityContext.allowPrivilegeEscalation Set primary container's Security Context allowPrivilegeEscalation -## @param containerSecurityContext.readOnlyRootFilesystem Set container's Security Context readOnlyRootFilesystem -## @param containerSecurityContext.capabilities.drop List of capabilities to be dropped -## @param containerSecurityContext.seccompProfile.type Set container's Security Context seccomp profile -## -containerSecurityContext: - enabled: true - seLinuxOptions: {} - runAsUser: 1001 - runAsGroup: 1001 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: ["ALL"] - seccompProfile: - type: "RuntimeDefault" -## etcd containers' resource requests and limits -## ref: https://kubernetes.io/docs/concepts/configuration/manage-compute-resources-container/ -## We usually recommend not to specify default resources and to leave this as a conscious -## choice for the user. This also increases chances charts run on environments with little -## resources, such as Minikube. If you do want to specify resources, uncomment the following -## lines, adjust them as necessary, and remove the curly braces after 'resources:'. -## @param resourcesPreset Set container resources according to one common preset (allowed values: none, nano, micro, small, medium, large, xlarge, 2xlarge). This is ignored if resources is set (resources is recommended for production). -## More information: https://github.com/bitnami/charts/blob/main/bitnami/common/templates/_resources.tpl#L15 -## -resourcesPreset: "micro" -## @param resources Set container requests and limits for different resources like CPU or memory (essential for production workloads) -## Example: -## resources: -## requests: -## cpu: 2 -## memory: 512Mi -## limits: -## cpu: 3 -## memory: 1024Mi -## -resources: {} -## Configure extra options for liveness probe -## ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/#configure-probes -## @param livenessProbe.enabled Enable livenessProbe -## @param livenessProbe.initialDelaySeconds Initial delay seconds for livenessProbe -## @param livenessProbe.periodSeconds Period seconds for livenessProbe -## @param livenessProbe.timeoutSeconds Timeout seconds for livenessProbe -## @param livenessProbe.failureThreshold Failure threshold for livenessProbe -## @param livenessProbe.successThreshold Success threshold for livenessProbe -## -livenessProbe: - enabled: true - initialDelaySeconds: 60 - periodSeconds: 30 - timeoutSeconds: 5 - successThreshold: 1 - failureThreshold: 5 -## Configure extra options for readiness probe -## ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/#configure-probes -## @param readinessProbe.enabled Enable readinessProbe -## @param readinessProbe.initialDelaySeconds Initial delay seconds for readinessProbe -## @param readinessProbe.periodSeconds Period seconds for readinessProbe -## @param readinessProbe.timeoutSeconds Timeout seconds for readinessProbe -## @param readinessProbe.failureThreshold Failure threshold for readinessProbe -## @param readinessProbe.successThreshold Success threshold for readinessProbe -## -readinessProbe: - enabled: true - initialDelaySeconds: 60 - periodSeconds: 10 - timeoutSeconds: 5 - successThreshold: 1 - failureThreshold: 5 -## Configure extra options for liveness probe -## ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/#configure-probes -## @param startupProbe.enabled Enable startupProbe -## @param startupProbe.initialDelaySeconds Initial delay seconds for startupProbe -## @param startupProbe.periodSeconds Period seconds for startupProbe -## @param startupProbe.timeoutSeconds Timeout seconds for startupProbe -## @param startupProbe.failureThreshold Failure threshold for startupProbe -## @param startupProbe.successThreshold Success threshold for startupProbe -## -startupProbe: - enabled: false - initialDelaySeconds: 0 - periodSeconds: 10 - timeoutSeconds: 5 - successThreshold: 1 - failureThreshold: 60 -## @param customLivenessProbe [object] Override default liveness probe -## -customLivenessProbe: {} -## @param customReadinessProbe [object] Override default readiness probe -## -customReadinessProbe: {} -## @param customStartupProbe [object] Override default startup probe -## -customStartupProbe: {} -## @param extraVolumes [array] Optionally specify extra list of additional volumes for etcd pods -## -extraVolumes: [] -## @param extraVolumeMounts [array] Optionally specify extra list of additional volumeMounts for etcd container(s) -## -extraVolumeMounts: [] -## @param extraVolumeClaimTemplates [array] Optionally specify extra list of additional volumeClaimTemplates for etcd container(s) -## -extraVolumeClaimTemplates: [] -## @param initContainers [array] Add additional init containers to the etcd pods -## e.g: -## initContainers: -## - name: your-image-name -## image: your-image -## imagePullPolicy: Always -## ports: -## - name: portname -## containerPort: 1234 -## -initContainers: [] -## @param sidecars [array] Add additional sidecar containers to the etcd pods -## e.g: -## sidecars: -## - name: your-image-name -## image: your-image -## imagePullPolicy: Always -## ports: -## - name: portname -## containerPort: 1234 -## -sidecars: [] -## @param podAnnotations [object] Annotations for etcd pods -## ref: https://kubernetes.io/docs/concepts/overview/working-with-objects/annotations/ -## -podAnnotations: {} -## @param podLabels [object] Extra labels for etcd pods -## Ref: https://kubernetes.io/docs/concepts/overview/working-with-objects/labels/ -## -podLabels: {} -## @param podAffinityPreset Pod affinity preset. Ignored if `affinity` is set. Allowed values: `soft` or `hard` -## ref: https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#inter-pod-affinity-and-anti-affinity -## -podAffinityPreset: "" -## @param podAntiAffinityPreset Pod anti-affinity preset. Ignored if `affinity` is set. Allowed values: `soft` or `hard` -## Ref: https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#inter-pod-affinity-and-anti-affinity -## -podAntiAffinityPreset: soft -## Node affinity preset -## Ref: https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#node-affinity -## @param nodeAffinityPreset.type Node affinity preset type. Ignored if `affinity` is set. Allowed values: `soft` or `hard` -## @param nodeAffinityPreset.key Node label key to match. Ignored if `affinity` is set. -## @param nodeAffinityPreset.values [array] Node label values to match. Ignored if `affinity` is set. -## -nodeAffinityPreset: - type: "" - ## e.g: - ## key: "kubernetes.io/e2e-az-name" - ## - key: "" - ## e.g: - ## values: - ## - e2e-az1 - ## - e2e-az2 - ## - values: [] -## @param affinity [object] Affinity for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## Note: podAffinityPreset, podAntiAffinityPreset, and nodeAffinityPreset will be ignored when it's set -## -affinity: {} -## @param nodeSelector [object] Node labels for pod assignment -## Ref: https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/ -## -nodeSelector: {} -## @param tolerations [array] Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -## -tolerations: [] -## @param terminationGracePeriodSeconds Seconds the pod needs to gracefully terminate -## ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/#hook-handler-execution -## -terminationGracePeriodSeconds: "" -## @param schedulerName Name of the k8s scheduler (other than default) -## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ -## -schedulerName: "" -## @param priorityClassName Name of the priority class to be used by etcd pods -## Priority class needs to be created beforehand -## Ref: https://kubernetes.io/docs/concepts/configuration/pod-priority-preemption/ -## -priorityClassName: "" -## @param runtimeClassName Name of the runtime class to be used by pod(s) -## ref: https://kubernetes.io/docs/concepts/containers/runtime-class/ -## -runtimeClassName: "" -## @param shareProcessNamespace Enable shared process namespace in a pod. -## If set to false (default), each container will run in separate namespace, etcd will have PID=1. -## If set to true, the /pause will run as init process and will reap any zombie PIDs, -## for example, generated by a custom exec probe running longer than a probe timeoutSeconds. -## Enable this only if customLivenessProbe or customReadinessProbe is used and zombie PIDs are accumulating. -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/share-process-namespace/ -## -shareProcessNamespace: false -## @param topologySpreadConstraints Topology Spread Constraints for pod assignment -## https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -## The value is evaluated as a template -## -topologySpreadConstraints: [] -## persistentVolumeClaimRetentionPolicy -## ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#persistentvolumeclaim-retention -## @param persistentVolumeClaimRetentionPolicy.enabled Controls if and how PVCs are deleted during the lifecycle of a StatefulSet -## @param persistentVolumeClaimRetentionPolicy.whenScaled Volume retention behavior when the replica count of the StatefulSet is reduced -## @param persistentVolumeClaimRetentionPolicy.whenDeleted Volume retention behavior that applies when the StatefulSet is deleted -persistentVolumeClaimRetentionPolicy: - enabled: false - whenScaled: Retain - whenDeleted: Retain -## @section Traffic exposure parameters -## - -service: - ## @param service.type Kubernetes Service type - ## - type: ClusterIP - ## @param service.enabled create second service if equal true - ## - enabled: true - ## @param service.clusterIP Kubernetes service Cluster IP - ## e.g.: - ## clusterIP: None - ## - clusterIP: "" - ## @param service.ports.client etcd client port - ## @param service.ports.peer etcd peer port - ## @param service.ports.metrics etcd metrics port when metrics.useSeparateEndpoint is true - ## - ports: - client: 2379 - peer: 2380 - metrics: 9090 - ## @param service.nodePorts.client Specify the nodePort client value for the LoadBalancer and NodePort service types. - ## @param service.nodePorts.peer Specify the nodePort peer value for the LoadBalancer and NodePort service types. - ## @param service.nodePorts.metrics Specify the nodePort metrics value for the LoadBalancer and NodePort service types. The metrics port is only exposed when metrics.useSeparateEndpoint is true. - ## ref: https://kubernetes.io/docs/concepts/services-networking/service/#type-nodeport - ## - nodePorts: - client: "" - peer: "" - metrics: "" - ## @param service.clientPortNameOverride etcd client port name override - ## - clientPortNameOverride: "" - ## @param service.peerPortNameOverride etcd peer port name override - ## - peerPortNameOverride: "" - ## @param service.metricsPortNameOverride etcd metrics port name override. The metrics port is only exposed when metrics.useSeparateEndpoint is true. - ## - metricsPortNameOverride: "" - ## @param service.loadBalancerIP loadBalancerIP for the etcd service (optional, cloud specific) - ## ref: https://kubernetes.io/docs/concepts/services-networking/service/#type-loadbalancer - ## - loadBalancerIP: "" - ## @param service.loadBalancerSourceRanges [array] Load Balancer source ranges - ## ref: https://kubernetes.io/docs/tasks/access-application-cluster/configure-cloud-provider-firewall/#restrict-access-for-loadbalancer-service - ## e.g: - ## loadBalancerSourceRanges: - ## - 10.10.10.0/24 - ## - loadBalancerSourceRanges: [] - ## @param service.externalIPs [array] External IPs - ## ref: https://kubernetes.io/docs/concepts/services-networking/service/#external-ips - ## - externalIPs: [] - ## @param service.externalTrafficPolicy %%MAIN_CONTAINER_NAME%% service external traffic policy - ## ref http://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - ## - externalTrafficPolicy: Cluster - ## @param service.extraPorts Extra ports to expose (normally used with the `sidecar` value) - ## - extraPorts: [] - ## @param service.annotations [object] Additional annotations for the etcd service - ## - annotations: {} - ## @param service.sessionAffinity Session Affinity for Kubernetes service, can be "None" or "ClientIP" - ## If "ClientIP", consecutive client requests will be directed to the same Pod - ## ref: https://kubernetes.io/docs/concepts/services-networking/service/#virtual-ips-and-service-proxies - ## - sessionAffinity: None - ## @param service.sessionAffinityConfig Additional settings for the sessionAffinity - ## sessionAffinityConfig: - ## clientIP: - ## timeoutSeconds: 300 - ## - sessionAffinityConfig: {} - ## Headless service properties - ## - headless: - ## @param service.headless.annotations Annotations for the headless service. - ## - annotations: {} -## @section Persistence parameters -## - -## Enable persistence using Persistent Volume Claims -## ref: https://kubernetes.io/docs/concepts/storage/persistent-volumes/ -## -persistence: - ## @param persistence.enabled If true, use a Persistent Volume Claim. If false, use emptyDir. - ## - enabled: true - ## @param persistence.storageClass Persistent Volume Storage Class - ## If defined, storageClassName: - ## If set to "-", storageClassName: "", which disables dynamic provisioning - ## If undefined (the default) or set to null, no storageClassName spec is - ## set, choosing the default provisioner. (gp2 on AWS, standard on - ## GKE, AWS & OpenStack) - ## - storageClass: "sc-pd-standard" - ## - ## @param persistence.annotations [object] Annotations for the PVC - ## - annotations: {} - ## @param persistence.labels [object] Labels for the PVC - ## - labels: {} - ## @param persistence.accessModes Persistent Volume Access Modes - ## - accessModes: - - ReadWriteOnce - ## @param persistence.size PVC Storage Request for etcd data volume - ## - size: 8Gi - ## @param persistence.selector [object] Selector to match an existing Persistent Volume - ## ref: https://kubernetes.io/docs/concepts/storage/persistent-volumes/#selector - ## - selector: {} -## @section Volume Permissions parameters -## - -## Init containers parameters: -## volumePermissions: Change the owner and group of the persistent volume mountpoint to runAsUser:fsGroup values from the securityContext section. -## -volumePermissions: - ## @param volumePermissions.enabled Enable init container that changes the owner and group of the persistent volume(s) mountpoint to `runAsUser:fsGroup` - ## - enabled: false - ## @param volumePermissions.image.registry [default: REGISTRY_NAME] Init container volume-permissions image registry - ## @param volumePermissions.image.repository [default: REPOSITORY_NAME/os-shell] Init container volume-permissions image name - ## @skip volumePermissions.image.tag Init container volume-permissions image tag - ## @param volumePermissions.image.digest Init container volume-permissions image digest in the way sha256:aa.... Please note this parameter, if set, will override the tag - ## - image: - registry: docker.io - repository: bitnami/os-shell - tag: 12-debian-12-r30 - digest: "" - ## @param volumePermissions.image.pullPolicy Init container volume-permissions image pull policy - ## - pullPolicy: IfNotPresent - ## @param volumePermissions.image.pullSecrets [array] Specify docker-registry secret names as an array - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## e.g: - ## pullSecrets: - ## - myRegistryKeySecretName - ## - pullSecrets: [] - ## Init container' resource requests and limits - ## ref: https://kubernetes.io/docs/concepts/configuration/manage-compute-resources-container/ - ## We usually recommend not to specify default resources and to leave this as a conscious - ## choice for the user. This also increases chances charts run on environments with little - ## resources, such as Minikube. If you do want to specify resources, uncomment the following - ## lines, adjust them as necessary, and remove the curly braces after 'resources:'. - ## @param volumePermissions.resourcesPreset Set container resources according to one common preset (allowed values: none, nano, micro, small, medium, large, xlarge, 2xlarge). This is ignored if volumePermissions.resources is set (volumePermissions.resources is recommended for production). - ## More information: https://github.com/bitnami/charts/blob/main/bitnami/common/templates/_resources.tpl#L15 - ## - resourcesPreset: "nano" - ## @param volumePermissions.resources Set container requests and limits for different resources like CPU or memory (essential for production workloads) - ## Example: - ## resources: - ## requests: - ## cpu: 2 - ## memory: 512Mi - ## limits: - ## cpu: 3 - ## memory: 1024Mi - ## - resources: {} -## @section Network Policy parameters -## ref: https://kubernetes.io/docs/concepts/services-networking/network-policies/ -## -networkPolicy: - ## @param networkPolicy.enabled Enable creation of NetworkPolicy resources - ## - enabled: true - ## @param networkPolicy.allowExternal Don't require client label for connections - ## When set to false, only pods with the correct client label will have network access to the ports - ## etcd is listening on. When true, etcd will accept connections from any source - ## (with the correct destination port). - ## - allowExternal: true - ## @param networkPolicy.allowExternalEgress Allow the pod to access any range of port and all destinations. - ## - allowExternalEgress: true - ## @param networkPolicy.extraIngress [array] Add extra ingress rules to the NetworkPolicy - ## e.g: - ## extraIngress: - ## - ports: - ## - port: 1234 - ## from: - ## - podSelector: - ## - matchLabels: - ## - role: frontend - ## - podSelector: - ## - matchExpressions: - ## - key: role - ## operator: In - ## values: - ## - frontend - ## - extraIngress: [] - ## @param networkPolicy.extraEgress [array] Add extra ingress rules to the NetworkPolicy - ## e.g: - ## extraEgress: - ## - ports: - ## - port: 1234 - ## to: - ## - podSelector: - ## - matchLabels: - ## - role: frontend - ## - podSelector: - ## - matchExpressions: - ## - key: role - ## operator: In - ## values: - ## - frontend - ## - extraEgress: [] - ## @param networkPolicy.ingressNSMatchLabels [object] Labels to match to allow traffic from other namespaces - ## @param networkPolicy.ingressNSPodMatchLabels [object] Pod labels to match to allow traffic from other namespaces - ## - ingressNSMatchLabels: {} - ingressNSPodMatchLabels: {} -## @section Metrics parameters -## -metrics: - ## @param metrics.enabled Expose etcd metrics - ## - enabled: false - ## @param metrics.useSeparateEndpoint Use a separate endpoint for exposing metrics - # - useSeparateEndpoint: false - ## @param metrics.podAnnotations [object] Annotations for the Prometheus metrics on etcd pods - ## - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "{{ .Values.metrics.useSeparateEndpoint | ternary .Values.containerPorts.metrics .Values.containerPorts.client }}" - ## Prometheus Service Monitor - ## ref: https://github.com/coreos/prometheus-operator/blob/master/Documentation/api.md#endpoint - ## - podMonitor: - ## @param metrics.podMonitor.enabled Create PodMonitor Resource for scraping metrics using PrometheusOperator - ## - enabled: false - ## @param metrics.podMonitor.namespace Namespace in which Prometheus is running - ## - namespace: monitoring - ## @param metrics.podMonitor.interval Specify the interval at which metrics should be scraped - ## - interval: 30s - ## @param metrics.podMonitor.scrapeTimeout Specify the timeout after which the scrape is ended - ## - scrapeTimeout: 30s - ## @param metrics.podMonitor.additionalLabels [object] Additional labels that can be used so PodMonitors will be discovered by Prometheus - ## ref: https://github.com/coreos/prometheus-operator/blob/master/Documentation/api.md#prometheusspec - ## - additionalLabels: {} - ## @param metrics.podMonitor.scheme Scheme to use for scraping - ## - scheme: http - ## @param metrics.podMonitor.tlsConfig [object] TLS configuration used for scrape endpoints used by Prometheus - ## ref: https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#tlsconfig - ## e.g: - ## tlsConfig: - ## ca: - ## secret: - ## name: existingSecretName - ## - tlsConfig: {} - ## @param metrics.podMonitor.relabelings [array] Prometheus relabeling rules - ## - relabelings: [] - ## Prometheus Operator PrometheusRule configuration - ## - prometheusRule: - ## @param metrics.prometheusRule.enabled Create a Prometheus Operator PrometheusRule (also requires `metrics.enabled` to be `true` and `metrics.prometheusRule.rules`) - ## - enabled: false - ## @param metrics.prometheusRule.namespace Namespace for the PrometheusRule Resource (defaults to the Release Namespace) - ## - namespace: "" - ## @param metrics.prometheusRule.additionalLabels Additional labels that can be used so PrometheusRule will be discovered by Prometheus - ## - additionalLabels: {} - ## @param metrics.prometheusRule.rules Prometheus Rule definitions - # - alert: ETCD has no leader - # annotations: - # summary: "ETCD has no leader" - # description: "pod {{`{{`}} $labels.pod {{`}}`}} state error, can't connect leader" - # for: 1m - # expr: etcd_server_has_leader == 0 - # labels: - # severity: critical - # group: PaaS - ## - rules: [] -## @section Snapshotting parameters -## - -## Start a new etcd cluster recovering the data from an existing snapshot before bootstrapping -## -startFromSnapshot: - ## @param startFromSnapshot.enabled Initialize new cluster recovering an existing snapshot - ## - enabled: false - ## @param startFromSnapshot.existingClaim Existing PVC containing the etcd snapshot - ## - existingClaim: "" - ## @param startFromSnapshot.snapshotFilename Snapshot filename - ## - snapshotFilename: "" -## Enable auto disaster recovery by periodically snapshotting the keyspace: -## - It creates a cronjob to periodically snapshotting the keyspace -## - It also creates a ReadWriteMany PVC to store the snapshots -## If the cluster permanently loses more than (N-1)/2 members, it tries to -## recover itself from the last available snapshot. -## -disasterRecovery: - ## @param disasterRecovery.enabled Enable auto disaster recovery by periodically snapshotting the keyspace - ## - enabled: false - cronjob: - ## @param disasterRecovery.cronjob.schedule Schedule in Cron format to save snapshots - ## See https://en.wikipedia.org/wiki/Cron - ## - schedule: "*/30 * * * *" - ## @param disasterRecovery.cronjob.historyLimit Number of successful finished jobs to retain - ## - historyLimit: 1 - ## @param disasterRecovery.cronjob.snapshotHistoryLimit Number of etcd snapshots to retain, tagged by date - ## - snapshotHistoryLimit: 1 - ## @param disasterRecovery.cronjob.snapshotsDir Directory to store snapshots - ## - snapshotsDir: "/snapshots" - ## @param disasterRecovery.cronjob.podAnnotations [object] Pod annotations for cronjob pods - ## ref: https://kubernetes.io/docs/concepts/overview/working-with-objects/annotations/ - ## - podAnnotations: {} - ## Configure resource requests and limits for snapshotter containers - ## ref: https://kubernetes.io/docs/concepts/configuration/manage-compute-resources-container/ - ## We usually recommend not to specify default resources and to leave this as a conscious - ## choice for the user. This also increases chances charts run on environments with little - ## resources, such as Minikube. If you do want to specify resources, uncomment the following - ## lines, adjust them as necessary, and remove the curly braces after 'resources:'. - ## @param disasterRecovery.cronjob.resourcesPreset Set container resources according to one common preset (allowed values: none, nano, micro, small, medium, large, xlarge, 2xlarge). This is ignored if disasterRecovery.cronjob.resources is set (disasterRecovery.cronjob.resources is recommended for production). - ## More information: https://github.com/bitnami/charts/blob/main/bitnami/common/templates/_resources.tpl#L15 - ## - resourcesPreset: "nano" - ## @param disasterRecovery.cronjob.resources Set container requests and limits for different resources like CPU or memory (essential for production workloads) - ## Example: - ## resources: - ## requests: - ## cpu: 2 - ## memory: 512Mi - ## limits: - ## cpu: 3 - ## memory: 1024Mi - ## - resources: {} - ## @param disasterRecovery.cronjob.nodeSelector Node labels for cronjob pods assignment - ## Ref: https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/ - ## - nodeSelector: {} - ## @param disasterRecovery.cronjob.tolerations Tolerations for cronjob pods assignment - ## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ - ## - tolerations: [] - ## @param disasterRecovery.cronjob.podLabels [object] Labels that will be added to pods created by cronjob - ## - podLabels: {} - ## @param disasterRecovery.cronjob.serviceAccountName Specifies the service account to use for disaster recovery cronjob - ## - serviceAccountName: "" - ## @param disasterRecovery.cronjob.command Override default snapshot container command (useful when you want to customize the snapshot logic) - ## - command: [] - ## - pvc: - ## @param disasterRecovery.pvc.existingClaim A manually managed Persistent Volume and Claim - ## If defined, PVC must be created manually before volume will be bound - ## The value is evaluated as a template, so, for example, the name can depend on .Release or .Chart - ## - existingClaim: "" - ## @param disasterRecovery.pvc.size PVC Storage Request - ## - size: 2Gi - ## @param disasterRecovery.pvc.storageClassName Storage Class for snapshots volume - ## - storageClassName: nfs - ## @param disasterRecovery.pvc.subPath Path within the volume from which to mount - ## Useful if snapshots should only be stored in a subdirectory of the volume - ## - subPath: "" -## @section Service account parameters -## -serviceAccount: - ## @param serviceAccount.create Enable/disable service account creation - ## - create: true - ## @param serviceAccount.name Name of the service account to create or use - ## - name: "" - ## @param serviceAccount.automountServiceAccountToken Enable/disable auto mounting of service account token - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-service-account/#use-the-default-service-account-to-access-the-api-server - ## - automountServiceAccountToken: false - ## @param serviceAccount.annotations [object] Additional annotations to be included on the service account - ## - annotations: {} - ## @param serviceAccount.labels [object] Additional labels to be included on the service account - ## - labels: {} -## @section Defragmentation parameters -## - -## Enable defragmentation by periodically rearranging fragmented data after history compaction. -## It creates a cronjob to periodically run the defragmentation command: -## etcdctl defrag [OPTIONS] -## See https://etcd.io/docs/latest/op-guide/maintenance/ -## -defrag: - ## @param defrag.enabled Enable automatic defragmentation. This is most effective when paired with auto compaction: consider setting "autoCompactionRetention > 0". - ## - enabled: false - cronjob: - ## @param defrag.cronjob.startingDeadlineSeconds Number of seconds representing the deadline for starting the job if it misses scheduled time for any reason - ## - startingDeadlineSeconds: "" - ## @param defrag.cronjob.schedule Schedule in Cron format to defrag (daily at midnight by default) - ## See https://en.wikipedia.org/wiki/Cron - ## - schedule: "0 0 * * *" - ## @param defrag.cronjob.concurrencyPolicy Set the cronjob parameter concurrencyPolicy - ## - concurrencyPolicy: Forbid - ## @param defrag.cronjob.suspend Boolean that indicates if the controller must suspend subsequent executions (not applied to already started executions) - ## - suspend: false - ## @param defrag.cronjob.successfulJobsHistoryLimit Number of successful finished jobs to retain - ## - successfulJobsHistoryLimit: 1 - ## @param defrag.cronjob.failedJobsHistoryLimit Number of failed finished jobs to retain - ## - failedJobsHistoryLimit: 1 - ## @param defrag.cronjob.labels [object] Additional labels to be added to the Defrag cronjob - ## - labels: {} - ## @param defrag.cronjob.annotations [object] Annotations to be added to the Defrag cronjob - ## - annotations: {} - ## @param defrag.cronjob.activeDeadlineSeconds Number of seconds relative to the startTime that the job may be continuously active before the system tries to terminate it - ## - activeDeadlineSeconds: "" - ## @param defrag.cronjob.restartPolicy Set the cronjob parameter restartPolicy - ## - restartPolicy: OnFailure - ## @param defrag.cronjob.podLabels [object] Labels that will be added to pods created by Defrag cronjob - ## - podLabels: {} - ## @param defrag.cronjob.podAnnotations [object] Pod annotations for Defrag cronjob pods - ## ref: https://kubernetes.io/docs/concepts/overview/working-with-objects/annotations/ - ## - podAnnotations: {} - ## K8s Security Context for Defrag cronjob pods - ## https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - ## @param defrag.cronjob.podSecurityContext.enabled Enable security context for Defrag pods - ## @param defrag.cronjob.podSecurityContext.fsGroupChangePolicy Set filesystem group change policy - ## @param defrag.cronjob.podSecurityContext.sysctls Set kernel settings using the sysctl interface - ## @param defrag.cronjob.podSecurityContext.supplementalGroups Set filesystem extra groups - ## @param defrag.cronjob.podSecurityContext.fsGroup Group ID for the Defrag filesystem - ## - podSecurityContext: - enabled: true - fsGroupChangePolicy: Always - sysctls: [] - supplementalGroups: [] - fsGroup: 1001 - ## Configure container security context for Defrag cronjob pods - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-pod - ## @param defrag.cronjob.containerSecurityContext.enabled Enabled containers' Security Context - ## @param defrag.cronjob.containerSecurityContext.seLinuxOptions [object,nullable] Set SELinux options in container - ## @param defrag.cronjob.containerSecurityContext.runAsUser Set containers' Security Context runAsUser - ## @param defrag.cronjob.containerSecurityContext.runAsGroup Set containers' Security Context runAsGroup - ## @param defrag.cronjob.containerSecurityContext.runAsNonRoot Set container's Security Context runAsNonRoot - ## @param defrag.cronjob.containerSecurityContext.privileged Set container's Security Context privileged - ## @param defrag.cronjob.containerSecurityContext.readOnlyRootFilesystem Set container's Security Context readOnlyRootFilesystem - ## @param defrag.cronjob.containerSecurityContext.allowPrivilegeEscalation Set container's Security Context allowPrivilegeEscalation - ## @param defrag.cronjob.containerSecurityContext.capabilities.drop List of capabilities to be dropped - ## @param defrag.cronjob.containerSecurityContext.seccompProfile.type Set container's Security Context seccomp profile - ## - containerSecurityContext: - enabled: true - seLinuxOptions: {} - runAsUser: 1001 - runAsGroup: 1001 - runAsNonRoot: true - privileged: false - readOnlyRootFilesystem: true - allowPrivilegeEscalation: false - capabilities: - drop: ["ALL"] - seccompProfile: - type: "RuntimeDefault" - ## @param defrag.cronjob.nodeSelector [object] Node labels for pod assignment in Defrag cronjob - ## Ref: https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/ - ## - nodeSelector: {} - ## @param defrag.cronjob.tolerations [array] Tolerations for pod assignment in Defrag cronjob - ## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ - ## - tolerations: [] - ## @param defrag.cronjob.serviceAccountName Specifies the service account to use for Defrag cronjob - ## - serviceAccountName: "" - ## @param defrag.cronjob.command [array] Override default container command for defragmentation (useful when using custom images) - ## ref: https://kubernetes.io/docs/concepts/overview/working-with-objects/annotations/ - ## - command: [] - ## @param defrag.cronjob.args [array] Override default container args (useful when using custom images) - ## - args: [] - ## @param defrag.cronjob.resourcesPreset Set container resources according to one common preset - ## (allowed values: none, nano, micro, small, medium, large, xlarge, 2xlarge). This is ignored if - ## defrag.cronjob.resources is set (defrag.cronjob.resources is recommended for production). - ## More information: https://github.com/bitnami/charts/blob/main/bitnami/common/templates/_resources.tpl#L15 - ## - resourcesPreset: "nano" - ## @param defrag.cronjob.resources [object] Set container requests and limits for different resources like CPU or - ## memory (essential for production workloads) - ## Example: - ## resources: - ## requests: - ## cpu: 2 - ## memory: 512Mi - ## limits: - ## cpu: 3 - ## memory: 1024Mi - ## - resources: {} -## @section Other parameters -## - -## etcd Pod Disruption Budget configuration -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -## -pdb: - ## @param pdb.create Enable/disable a Pod Disruption Budget creation - ## - create: true - ## @param pdb.minAvailable Minimum number/percentage of pods that should remain scheduled - ## - minAvailable: 51% - ## @param pdb.maxUnavailable Maximum number/percentage of pods that may be made unavailable - ## - maxUnavailable: "" - - -#httpproxy values -httpProxy: - enabled: true - ingressClassName: contour-internal-1 - virtualhost: etcd-central.int.meesho.int diff --git a/helm-overrides/gke-central-prd-ase1a/external-secrets/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/external-secrets/custom-values.yaml deleted file mode 100644 index 5fdea83..0000000 --- a/helm-overrides/gke-central-prd-ase1a/external-secrets/custom-values.yaml +++ /dev/null @@ -1,62 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - - # -- Annotations to add to Pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: central-devops - nodeSelector: - cloud.google.com/compute-class: central-devops - certController: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: central-devops - nodeSelector: - cloud.google.com/compute-class: central-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - webhook: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: central-devops - nodeSelector: - cloud.google.com/compute-class: central-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" diff --git a/helm-overrides/gke-central-prd-ase1a/fireworks-ai/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/fireworks-ai/custom-values.yaml deleted file mode 100644 index 3d47199..0000000 --- a/helm-overrides/gke-central-prd-ase1a/fireworks-ai/custom-values.yaml +++ /dev/null @@ -1,279 +0,0 @@ -# Central production values for the Bifrost 1.5.12 upgrade candidate. -# Chart: helm-templates/bifrost-v1.5.12 - -replicaCount: 1 - -fullnameOverride: "fireworks-ai" - -image: - repository: docker.io/maximhq/bifrost - pullPolicy: IfNotPresent - tag: "v1.5.12" - -serviceAccount: - create: true - automount: true - annotations: - iam.gke.io/gcp-service-account: sa-dvops-ai-gateway@meesho-central-prd-0622.iam.gserviceaccount.com - name: "fireworks-ai" - -deploymentLabels: - bu: central - env: prod - team: devops - priority: p1 - priority_v2: sp1 - primary_owner: deep.shah - secondary_owner: ramiz.mehran - service: fireworks-ai - service_type: producer-httpstateless - -podLabels: - bu: central - env: prod - team: devops - priority: p1 - priority_v2: sp1 - primary_owner: deep.shah - secondary_owner: ramiz.mehran - service: fireworks-ai - service_type: producer-httpstateless - -podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: "8080" - prometheus.io/scrape: "true" - telegraf.influxdata.com/class: infra - -podSecurityContext: - fsGroup: 65534 - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - -securityContext: - capabilities: - drop: - - ALL - readOnlyRootFilesystem: false - runAsNonRoot: true - runAsUser: 65534 - -service: - type: ClusterIP - port: 8080 - -httpProxy: - enabled: true -createContourGateway: true -namespace: prd-fireworks-ai -contourResponseTimeout: false -ingress: - enabled: false - ingressClassName: contour-internal-1 - servicePortNumber: 8080 - enableWebsocket: false - hosts: - - host: fireworks-ai.prd.meesho.int - paths: - - path: / - pathType: ImplementationSpecific - slowStart: - enabled: false - aggression: 1 - minPercent: 10 - window: 120s - -resources: - limits: - cpu: "4" - memory: 8Gi - requests: - cpu: "1" - memory: 2Gi - -livenessProbe: - httpGet: - path: /health - port: http - scheme: HTTP - initialDelaySeconds: 15 - periodSeconds: 5 - timeoutSeconds: 3 - failureThreshold: 5 - successThreshold: 1 - -readinessProbe: - httpGet: - path: /health - port: http - scheme: HTTP - initialDelaySeconds: 15 - periodSeconds: 5 - timeoutSeconds: 3 - failureThreshold: 5 - successThreshold: 1 - -autoscaling: - enabled: false - -nodeSelector: - cloud.google.com/compute-class: megatetralite - -tolerations: - - key: cloud.google.com/compute-class - operator: Equal - value: megatetralite - effect: NoSchedule - -affinity: {} - -strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 100% - maxUnavailable: 0 - -terminationGracePeriodSeconds: 300 -lifecycle: - preStop: - exec: - command: - - /bin/sh - - -c - - sleep 120 - -bifrost: - appDir: /app/data - port: 8080 - host: 0.0.0.0 - logLevel: warn - logStyle: json - - client: - dropExcessRequests: false - initialPoolSize: 300 - allowedOrigins: - - "*" - enableLogging: true - disableContentLogging: false - disableDbPingsInHealth: false - dumpErrorsInConsoleLogs: false - logRetentionDays: 365 - enforceGovernanceHeader: true - enforceAuthOnInference: false - allowDirectKeys: false - maxRequestBodySizeMb: 100 - -storage: - mode: postgres - configStore: - enabled: true - logsStore: - enabled: true - -# PostgreSQL is managed by this Helm release as a separate Deployment. -postgresql: - enabled: true - external: - enabled: false - image: - repository: docker.io/library/postgres - tag: "16-alpine" - pullPolicy: IfNotPresent - auth: - username: bifrost - database: bifrost - existingSecret: fireworks-ai-vault - passwordKey: BIFROST_POSTGRES_PASSWORD - primary: - persistence: - enabled: true - size: 100Gi - storageClass: hyperdisk-balanced - resources: - limits: - cpu: "1" - memory: 2Gi - requests: - cpu: 250m - memory: 512Mi - podSecurityContext: - fsGroup: 999 - containerSecurityContext: {} - nodeSelector: - cloud.google.com/compute-class: megatetralite - tolerations: - - key: cloud.google.com/compute-class - operator: Equal - value: megatetralite - effect: NoSchedule - affinity: {} - -vectorStore: - enabled: false - type: none - -env: - - name: TZ - value: "Asia/Kolkata" - - name: TELEGRAF_UDP_HOST - valueFrom: - fieldRef: - fieldPath: status.podIP - - name: NODE_IP - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - name: POD_NAME - valueFrom: - fieldRef: - fieldPath: metadata.name - - name: POD_NAMESPACE - valueFrom: - fieldRef: - fieldPath: metadata.namespace - - name: POD_IP - valueFrom: - fieldRef: - fieldPath: status.podIP - -podDisruptionBudget: - enabled: true - maxUnavailable: "10%" - -# Reuses the current production Vault path until a dedicated path is provisioned. -externalSecret: - enabled: true - secretName: fireworks-ai-vault - path: "prd/cntr/devop/ai-gateway" - refreshInterval: "0" - secretStoreRef: vault-backend - -keda: - enabled: true - pollingInterval: 30 - minReplicaCount: 2 - maxReplicaCount: 200 - scaledown: - stabilizationWindowSeconds: 1800 - selectpolicy: Min - policies: - - type: Pods - value: 2 - periodseconds: 15 - scaleup: - stabilizationWindowSeconds: 120 - selectpolicy: Max - policies: - - type: Pods - value: 2 - periodseconds: 15 - - type: Percent - value: 10 - periodseconds: 15 - triggers: - - type: cpu - metricType: Utilization - metadata: - value: "40" diff --git a/helm-overrides/gke-central-prd-ase1a/flagger/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/flagger/custom-values.yaml deleted file mode 100644 index 0e46d4a..0000000 --- a/helm-overrides/gke-central-prd-ase1a/flagger/custom-values.yaml +++ /dev/null @@ -1,68 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - cloud.google.com/compute-class: central-devops - -tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: central-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: central-devops - bu: infra - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-central-rollout-service.prd-central-rollout-service.svc.cluster.local" - - -rolloutService: - enabled: true - apps: - prd-devops-node-app-consumers: true - prd-devops-node-app: true - prd-devops-golang-app: true - prd-devops-golang-app-test: true - prd-devops-golang-app-crons: true diff --git a/helm-overrides/gke-central-prd-ase1a/fluentd-copy/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/fluentd-copy/custom-values.yaml deleted file mode 100644 index 58bdf35..0000000 --- a/helm-overrides/gke-central-prd-ase1a/fluentd-copy/custom-values.yaml +++ /dev/null @@ -1,787 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-cntr-cnsre-fluentd-prd@meesho-central-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: node_pool - operator: In - values: - - np-cntr-cndvs-blue-amd-prd-ase1 - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: central - team: sre - type: fluentd - service: fluentd-copy-central-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: central - team: sre - type: fluentd - service: fluentd-copy-central-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: debug -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: [] -# - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/gke-central-prd-ase1a/fluentd/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/fluentd/custom-values.yaml deleted file mode 100644 index 76df75d..0000000 --- a/helm-overrides/gke-central-prd-ase1a/fluentd/custom-values.yaml +++ /dev/null @@ -1,829 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-cntr-cnsre-fluentd-prd@meesho-central-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: node_pool - operator: NotIn - values: - - np-cntr-cndvs-blue-amd-prd-ase1 - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: central - team: sre - type: fluentd - service: fluentd-central-a-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: central - team: sre - type: fluentd - service: fluentd-central-a-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -# - secretRef: -# name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: - - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - from_encoding "utf-8" - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/gke-central-prd-ase1a/keda/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/keda/custom-values.yaml deleted file mode 100644 index c705e88..0000000 --- a/helm-overrides/gke-central-prd-ase1a/keda/custom-values.yaml +++ /dev/null @@ -1,42 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - cloud.google.com/compute-class: central-devops - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - podLabels: - bu: "central" - team: "central-devops" - metricsAdapter: - bu: "central" - team: "central-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 200m - memory: 200Mi - metricServer: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 200m - memory: 200Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 50m - memory: 150Mi - requests: - cpu: 10m - memory: 80Mi diff --git a/helm-overrides/gke-central-prd-ase1a/kube-dns/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/kube-dns/custom-values.yaml deleted file mode 100644 index 56e51d5..0000000 --- a/helm-overrides/gke-central-prd-ase1a/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.1.80.2"],"prd.mrouter.int.svc.cluster.local":["10.1.80.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.1.80.2"]} diff --git a/helm-overrides/gke-central-prd-ase1a/kube-events/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/kube-events/custom-values.yaml deleted file mode 100644 index 6244914..0000000 --- a/helm-overrides/gke-central-prd-ase1a/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-central-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - cloud.google.com/compute-class: central-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: central-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: central-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: central-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: central-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: central-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/gke-central-prd-ase1a/kube-state-metrics/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 2d93b91..0000000 --- a/helm-overrides/gke-central-prd-ase1a/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-central-a-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-central-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "central" - team: "central-sre" - service: "kube-state-metrics-central-a-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 500m - memory: 1.5Gi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/gke-central-prd-ase1a/kubectl-mcp-server/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/kubectl-mcp-server/custom-values.yaml deleted file mode 100644 index 96d3485..0000000 --- a/helm-overrides/gke-central-prd-ase1a/kubectl-mcp-server/custom-values.yaml +++ /dev/null @@ -1,144 +0,0 @@ -fullnameOverride: "kubectl-mcp-server" - -replicas: 1 - -image: - # In-house hardened build (helm-templates/kubectl-mcp-server/docker/) — - # NOT the upstream Docker Hub image. 266→18 HIGH/CRIT vulns, non-root, - # multi-stage slim base, kubectl v1.33.12 / helm v3.21.0. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/prd/devop/kubectl-mcp-server - tag: "v2" - pullPolicy: IfNotPresent - -labels: - bu: infra - team: devops - service: kubectl-mcp-server - env: prd - -serviceAccount: - create: true - annotations: {} - -rbac: - create: true - -podAnnotations: {} - -podSecurityContext: {} - -securityContext: - runAsNonRoot: true - runAsUser: 1000 - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -priorityClassName: "" - -nodeSelector: - cloud.google.com/compute-class: "central-devops" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "central-devops" - effect: NoSchedule - -affinity: {} - -resources: - requests: - cpu: 250m - memory: 512Mi - limits: - cpu: 1000m - memory: 1Gi - -mcp: - mode: multi - transport: http - host: "0.0.0.0" - port: 8000 - -auth: - allowAnonymous: false - -externalSecrets: - enabled: true - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/prd/cntr/devop/kubectl-mcp-server" - -federation: - forwardTimeout: 30 - clusters: - - name: self - self: true - - name: admin - endpoint: http://kubectl-mcp-server-admin.prd.meesho.int/mcp - tokenEnv: ADMIN_MCP_TOKEN - - name: dataengg - endpoint: http://kubectl-mcp-server-dataengg.prd.meesho.int/mcp - tokenEnv: DATAENGG_MCP_TOKEN - - name: datascience - endpoint: http://kubectl-mcp-server-datascience.prd.meesho.int/mcp - tokenEnv: DATASCIENCE_MCP_TOKEN - - name: demand - endpoint: http://kubectl-mcp-server-demand.prd.meesho.int/mcp - tokenEnv: DEMAND_MCP_TOKEN - - name: dsgpu - endpoint: http://kubectl-mcp-server-dsgpu.prd.meesho.int/mcp - tokenEnv: DSGPU_MCP_TOKEN - - name: farmiso - endpoint: http://kubectl-mcp-server-farmiso.prd.meesho.int/mcp - tokenEnv: FARMISO_MCP_TOKEN - - name: supply - endpoint: http://kubectl-mcp-server-supply.prd.meesho.int/mcp - tokenEnv: SUPPLY_MCP_TOKEN - -# tcpSocket on purpose — streamable-http transport exposes only /mcp, -# there is no /health route (see chart values.yaml note). -livenessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - -readinessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - -service: - port: 8000 - type: ClusterIP - -# Contour standard (2.0.0-style): renders HTTPProxy parent/child + -intra -# instead of a networking.k8s.io Ingress. -createContourGateway: true - -ingress: - enabled: true - ingressClassName: contour-internal-1 - servicePortNumber: 8000 - hosts: - - host: kubectl-mcp-server-central.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 diff --git a/helm-overrides/gke-central-prd-ase1a/kyverno/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/kyverno/custom-values.yaml deleted file mode 100644 index fccbbc6..0000000 --- a/helm-overrides/gke-central-prd-ase1a/kyverno/custom-values.yaml +++ /dev/null @@ -1,2244 +0,0 @@ -global: - - # -- Internal settings used with `helm template` to generate install manifest - # @ignored - templating: - enabled: false - debug: false - version: ~ - - image: - # -- (string) Global value that allows to set a single image registry across all deployments. - # When set, it will override any values set under `.image.registry` across the chart. - registry: ~ - # -- (list) Global list of Image pull secrets - # When set, it will override any values set under `imagePullSecrets` under different components across the chart. - imagePullSecrets: [] - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - caCertificates: - # -- Global CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - # Individual controller values will override this global value - data: ~ - - # -- Global value to set single volume to be mounted for CA certificates for all deployments. - # Not used when `.Values.global.caCertificates.data` is defined - # Individual controller values will override this global value - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Additional container environment variables to apply to all containers and init containers - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Global node labels for pod assignment. Non-global values will override the global value. - nodeSelector: - cloud.google.com/compute-class: central-kyverno - - # -- Global List of node taints to tolerate. Non-global values will override the global value. - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "central-kyverno" - effect: "NoSchedule" - -# -- (string) Override the name of the chart -nameOverride: ~ - -# -- (string) Override the expanded name of the chart -fullnameOverride: ~ - -# -- (string) Override the namespace the chart deploys to -namespaceOverride: ~ - -upgrade: - # -- Upgrading from v2 to v3 is not allowed by default, set this to true once changes have been reviewed. - fromV2: false - -apiVersionOverride: - # -- (string) Override api version used to create `PodDisruptionBudget`` resources. - # When not specified the chart will check if `policy/v1/PodDisruptionBudget` is available to - # determine the api version automatically. - podDisruptionBudget: ~ - -rbac: - roles: - # -- Aggregate ClusterRoles to Kubernetes default user-facing roles. For more information, see [User-facing roles](https://kubernetes.io/docs/reference/access-authn-authz/rbac/#user-facing-roles) - aggregate: - admin: true - view: true - -# Use openreports.io as the API group for reporting -openreports: - # -- Enable OpenReports feature in controllers - enabled: false - # -- Whether to install CRDs from the upstream OpenReports chart. Setting this to true requires enabled to also be true. - installCrds: false - -# CRDs configuration -crds: - - # -- Whether to have Helm install the Kyverno CRDs, if the CRDs are not installed by Helm, they must be added before policies can be created - install: true - - reportsServer: - # -- Kyverno reports-server is used in your cluster - enabled: false - - groups: - - # -- Install CRDs in group `kyverno.io` - kyverno: - cleanuppolicies: true - clustercleanuppolicies: true - clusterpolicies: true - globalcontextentries: true - policies: true - policyexceptions: true - updaterequests: true - - # -- Install CRDs in group `policies.kyverno.io` - policies: - validatingpolicies: true - policyexceptions: true - imagevalidatingpolicies: true - namespacedimagevalidatingpolicies: true - mutatingpolicies: true - generatingpolicies: true - deletingpolicies: true - namespaceddeletingpolicies: true - namespacedvalidatingpolicies: true - - # -- Install CRDs in group `reports.kyverno.io` - reports: - clusterephemeralreports: true - ephemeralreports: true - - # -- Install CRDs in group `wgpolicyk8s.io` - wgpolicyk8s: - clusterpolicyreports: false ##lk - policyreports: false ##lk - - # -- Additional CRDs annotations (Replace=true avoids 262144-byte annotation limit on large CRDs) - annotations: - argocd.argoproj.io/sync-options: Replace=true - # strategy.spinnaker.io/replace: 'true' - - # -- Additional CRDs labels - customLabels: {} - - migration: - - # -- Enable CRDs migration using helm post upgrade hook - enabled: true - - # -- Resources to migrate - resources: - - cleanuppolicies.kyverno.io - - clustercleanuppolicies.kyverno.io - - clusterpolicies.kyverno.io - - globalcontextentries.kyverno.io - - policies.kyverno.io - - policyexceptions.kyverno.io - - updaterequests.kyverno.io - - deletingpolicies.policies.kyverno.io - - generatingpolicies.policies.kyverno.io - - imagevalidatingpolicies.policies.kyverno.io - - namespacedimagevalidatingpolicies.policies.kyverno.io - - mutatingpolicies.policies.kyverno.io - - namespaceddeletingpolicies.policies.kyverno.io - - namespacedvalidatingpolicies.policies.kyverno.io - - policyexceptions.policies.kyverno.io - - validatingpolicies.policies.kyverno.io - - image: - # -- (string) Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- (string) Image repository - repository: kyverno/kyverno-cli - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- (string) Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podResources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -# Configuration -config: - - # -- Create the configmap. - create: true - - # -- Preserve the configmap settings during upgrade. - preserve: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - # -- Enable registry mutation for container images. Enabled by default. - enableDefaultRegistryMutation: true - - # -- The registry hostname used for the image mutation. - defaultRegistry: docker.io - - # -- Exclude groups - excludeGroups: - - system:nodes - - # -- Exclude usernames - excludeUsernames: [] - # - '!system:kube-scheduler' - - # -- Exclude roles - excludeRoles: [] - - # -- Exclude roles - excludeClusterRoles: [] - - # -- Generate success events. - generateSuccessEvents: false - - # -- Resource types to be skipped by the Kyverno policy engine. - # Make sure to surround each entry in quotes so that it doesn't get parsed as a nested YAML list. - # These are joined together without spaces, run through `tpl`, and the result is set in the config map. - # @default -- See [values.yaml](values.yaml) - resourceFilters: - - '[Event,*,*]' - # - '[*/*,kube-system,*]' - - '[*/*,kube-public,*]' - - '[*/*,kube-node-lease,*]' - - '[Node,*,*]' - - '[Node/?*,*,*]' - - '[APIService,*,*]' - - '[APIService/?*,*,*]' - - '[TokenReview,*,*]' - - '[SubjectAccessReview,*,*]' - - '[SelfSubjectAccessReview,*,*]' - - '[Binding,*,*]' - - '[Pod/binding,*,*]' - - '[ReplicaSet,*,*]' - - '[ReplicaSet/?*,*,*]' - - '[EphemeralReport,*,*]' - - '[ClusterEphemeralReport,*,*]' - # exclude resources from the chart - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:additional]' - - '[ClusterRoleBinding,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.configMapName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.metricsConfigMapName" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Job,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[Job/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.admission-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.background-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - # -- Sets the threshold for the total number of UpdateRequests generated for mutateExisitng and generate policies. - updateRequestThreshold: 1000 - - # -- Defines the `namespaceSelector`/`objectSelector` in the webhook configurations. - # The Kyverno namespace is excluded if `excludeKyvernoNamespace` is `true` (default) - webhooks: - # Exclude namespaces - # namespaceSelector: lk-s2 - # matchExpressions: - # - key: kubernetes.io/metadata.name - # operator: NotIn - # values: - # - kube-systemm lk-s2 - # Exclude objects - # objectSelector: - # matchExpressions: - # - key: webhooks.kyverno.io/exclude - # operator: DoesNotExist - - # -- Defines annotations to set on webhook configurations. - webhookAnnotations: - # Example to disable admission enforcer on AKS: - # 'admissions.enforcer/disabled': 'true' lk-s3 - - # -- Defines labels to set on webhook configurations. - webhookLabels: {} - # Example to adopt webhook resources in ArgoCD: - # 'argocd.argoproj.io/instance': 'kyverno' - - # -- Defines match conditions to set on webhook configurations (requires Kubernetes 1.27+). - matchConditions: [] - - # -- Exclude Kyverno namespace - # Determines if default Kyverno namespace exclusion is enabled for webhooks and resourceFilters - excludeKyvernoNamespace: true - - # -- resourceFilter namespace exclude - # Namespaces to exclude from the default resourceFilters - resourceFiltersExcludeNamespaces: [] - - # -- resourceFilters exclude list - # Items to exclude from config.resourceFilters - resourceFiltersExclude: [] - - # -- resourceFilter namespace include - # Namespaces to include to the default resourceFilters - resourceFiltersIncludeNamespaces: [] - - # -- resourceFilters include list - # Items to include to config.resourceFilters - resourceFiltersInclude: [] - -# Metrics configuration -metricsConfig: - - # -- Create the configmap. - create: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - namespaces: - - # -- List of namespaces to capture metrics for. - include: [] - - # -- list of namespaces to NOT capture metrics for. - exclude: [] - - # -- (string) Rate at which metrics should reset so as to clean up the memory footprint of kyverno metrics, if you might be expecting high memory footprint of Kyverno's metrics. Default: 0, no refresh of metrics. WARNING: This flag is not working since Kyverno 1.8.0 - metricsRefreshInterval: ~ - # metricsRefreshInterval: 24h - - # -- (list) Configures the bucket boundaries for all Histogram metrics, changing this configuration requires restart of the kyverno admission controller - bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 15, 20, 25, 30] - - # -- (map) Configures the exposure of individual metrics, by default all metrics and all labels are exported, changing this configuration requires restart of the kyverno admission controller - metricsExposure: - kyverno_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_image_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_mutating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_generating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_admission_review_duration_seconds: - # enabled: false - disabledLabelDimensions: ["resource_namespace"] - kyverno_policy_rule_info_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_policy_results_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_admission_requests_total: - disabledLabelDimensions: ["resource_namespace"] - kyverno_cleanup_controller_deletedobjects_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - -# -- Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -imagePullSecrets: {} - # regcred: - # registry: foo.example.com - # username: foobar - # password: secret - # regcred2: - # registry: bar.example.com - # username: barbaz - # password: secret2 - -# -- Existing Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -existingImagePullSecrets: [] - # - test-registry - # - other-test-registry - -# Tests configuration -test: - # -- Sleep time before running test - sleep: 20 - - image: - # -- (string) Image registry - registry: curlimages - # -- Image repository - repository: curl - # -- Image tag - # Defaults to `latest` if omitted - tag: '8.10.1' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - # -- Security context for the test containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- Additional Pod annotations - podAnnotations: {} - - # -- List of node taints to tolerate - tolerations: [] - -# -- Additional labels -customLabels: {} - - -webhooksCleanup: - # -- Create a helm pre-delete hook to cleanup webhooks. - enabled: true - - autoDeleteWebhooks: - # -- Allow webhooks controller to delete webhooks using finalizers - enabled: false - - image: - # -- (string) Image registry - registry: registry.k8s.io - # -- Image repository - repository: kubectl - # -- Image tag - # Defaults to `latest` if omitted - tag: 'v1.32.7' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -grafana: - # -- Enable grafana dashboard creation. - enabled: false - - # -- Configmap name template. - configMapName: '{{ include "kyverno.fullname" . }}-grafana' - - # -- (string) Namespace to create the grafana dashboard configmap. - # If not set, it will be created in the same namespace where the chart is deployed. - namespace: ~ - - # -- Grafana dashboard configmap annotations. - annotations: {} - - # -- Grafana dashboard configmap labels - labels: - grafana_dashboard: "1" - - # -- create GrafanaDashboard custom resource referencing to the configMap. - # according to https://grafana-operator.github.io/grafana-operator/docs/examples/dashboard_from_configmap/readme/ - grafanaDashboard: - create: false - folder: kyverno - allowCrossNamespaceImport: true - matchLabels: - dashboards: "grafana" - -# Features configuration -features: - admissionReports: - # -- Enables the feature - enabled: true - aggregateReports: - # -- Enables the feature - enabled: true - policyReports: - # -- Enables the feature - enabled: false ##lk - validatingAdmissionPolicyReports: - # -- Enables the feature - enabled: true - mutatingAdmissionPolicyReports: - # -- Enables the feature - enabled: false - reporting: - # -- Enables the feature - validate: true - # -- Enables the feature - mutate: true - # -- Enables the feature - mutateExisting: true - # -- Enables the feature - imageVerify: true - # -- Enables the feature - generate: true - autoUpdateWebhooks: - # -- Enables the feature - enabled: true - backgroundScan: - # -- Enables the feature - enabled: true - # -- Number of background scan workers - backgroundScanWorkers: 2 - # -- Background scan interval - backgroundScanInterval: 1h - # -- Skips resource filters in background scan - skipResourceFilters: true - configMapCaching: - # -- Enables the feature - enabled: true - controllerRuntimeMetrics: - # -- Bind address for controller-runtime metrics (use "0" to disable it) - bindAddress: ":8080" - deferredLoading: - # -- Enables the feature - enabled: true - dumpPayload: - # -- Enables the feature - enabled: false - forceFailurePolicyIgnore: - # -- Enables the feature - enabled: false - generateValidatingAdmissionPolicy: - # -- Enables the feature - enabled: true - generateMutatingAdmissionPolicy: - # -- Enables the feature - enabled: false - dumpPatches: - # -- Enables the feature - enabled: false - globalContext: - # -- Maximum allowed response size from API Calls. A value of 0 bypasses checks (not recommended) - maxApiCallResponseLength: 2000000 - logging: - # -- Logging format - format: text - # -- Logging verbosity - verbosity: 2 - omitEvents: - # -- Events which should not be emitted (possible values `PolicyViolation`, `PolicyApplied`, `PolicyError`, and `PolicySkipped`) - eventTypes: - - PolicyApplied - - PolicySkipped - # - PolicyViolation - # - PolicyError - policyExceptions: - # -- Enables the feature - enabled: true - # -- Restrict policy exceptions to a single namespace - # Set to "*" to allow exceptions in all namespaces - namespace: '' - protectManagedResources: - # -- Enables the feature - enabled: false - registryClient: - # -- Allow insecure registry - allowInsecure: false - # -- Enable registry client helpers - credentialHelpers: - - default - - google - - amazon - - azure - - github - ttlController: - # -- Reconciliation interval for the label based cleanup manager - reconciliationInterval: 1m - tuf: - # -- Enables the feature - enabled: false - # -- (string) Path to Tuf root - root: ~ - # -- (string) Raw Tuf root - rootRaw: ~ - # -- (string) Tuf mirror - mirror: ~ - -# Admission controller configuration -admissionController: - autoscaling: - # -- Enable horizontal pod autoscaling - enabled: true - - # -- Minimum number of pods - minReplicas: 1 - - # -- Maximum number of pods - maxReplicas: 10 - - # -- Target CPU utilization percentage - targetCPUUtilizationPercentage: 80 - - # -- Configurable scaling behavior - behavior: {} - - # -- Overrides features defined at the root level - featuresOverride: - admissionReports: - # -- Max number of admission reports allowed in flight until the admission controller stops creating new ones - backPressureThreshold: 1000 - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- The ServiceAccount name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno admission controller activities. - # This will help ensure Kyverno stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- admissionController webhook server port - # in case you are using hostNetwork: true, you might want to change the port the webhookServer is listening to - webhookServer: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - admission-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.admissionController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - initContainer: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyvernopre - # -- (string) Image tag - # If missing, defaults to image.tag - tag: ~ - # -- (string) Image pull policy - # If missing, defaults to image.pullPolicy - pullPolicy: ~ - - resources: - # -- Pod resource limits - limits: - cpu: 1000m - memory: 1Gi - # -- Pod resource requests - requests: - cpu: 1000m - memory: 1Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - container: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyverno - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - resources: - # -- Pod resource limits - limits: - cpu: 1900m - memory: 3Gi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Array of extra init containers - extraInitContainers: [] - # - name: init-container - # image: busybox - # command: ['sh', '-c', 'echo Hello'] - - # -- Array of extra containers to run alongside kyverno - extraContainers: [] - # - name: myapp-container - # image: busybox - # command: ['sh', '-c', 'echo Hello && sleep 3600'] - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Kyverno's metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Background controller configuration -backgroundController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable background controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: - - apiGroups: - - networking.k8s.io - resources: - - ingresses - - ingressclasses - - networkpolicies - verbs: - - create - - update - - patch - - delete - - apiGroups: - - rbac.authorization.k8s.io - resources: - - rolebindings - - roles - verbs: - - create - - update - - patch - - delete - - apiGroups: - - '' - resources: - - configmaps - - resourcequotas - - limitranges - verbs: - - create - - update - - patch - - delete - - apiGroups: - - resource.k8s.io - resources: - - resourceclaims - - resourceclaimtemplates - verbs: - - create - - delete - - update - - patch - - deletecollection - - apiGroups: - - autoscaling - resources: - - horizontalpodautoscalers - verbs: - - get - - list - - watch - - update - - patch - - apiGroups: - - apps - resources: - - deployments - - statefulsets - - replicasets - verbs: - - get - - list - - watch - - update - - patch - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - # - patch - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/background-controller - # -- Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 1900m - memory: 3Gi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - background-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.backgroundController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - # -- backgroundController server port - # in case you are using hostNetwork: true, you might want to change the port the backgroundController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Cleanup controller configuration -cleanupController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable cleanup controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - delete - # - list - # - watch - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/cleanup-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- cleanupController server port - # in case you are using hostNetwork: true, you might want to change the port the cleanupController is listening to - server: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 1100m - memory: 1.2Gi - # -- Pod resource requests - requests: - cpu: 1100m - memory: 1.2Gi - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - cleanup-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `service.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Reports controller configuration -reportsController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable reports controller. - enabled: false ##lk - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/reports-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno reports controller activities. - # This will help ensure Kyverno reports stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 200m - memory: 128Mi - # -- Pod resource requests - requests: - cpu: 100m - memory: 64Mi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - reports-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.reportsController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- (string) Service node port. - # Only used if `type` is `NodePort`. - nodePort: ~ - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- (string) Traces receiver address - address: ~ - # -- (string) Traces receiver port - port: ~ - # -- (string) Traces receiver credentials - creds: ~ - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- (string) Otel collector endpoint - collector: ~ - # -- (string) Otel collector credentials - creds: ~ - - # -- reportsController server port - # in case you are using hostNetwork: true, you might want to change the port the reportsController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - - # -- Enable sanity check for reports CRDs - sanityChecks: true diff --git a/helm-overrides/gke-central-prd-ase1a/kyverno/policies/protect-namespace.yaml b/helm-overrides/gke-central-prd-ase1a/kyverno/policies/protect-namespace.yaml deleted file mode 100644 index 54f5162..0000000 --- a/helm-overrides/gke-central-prd-ase1a/kyverno/policies/protect-namespace.yaml +++ /dev/null @@ -1,40 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: protect-critical-namespaces -spec: - validationFailureAction: Enforce - background: false - rules: - - name: deny-deletion-of-critical-namespaces - match: - any: - - resources: - kinds: - - Namespace - names: - - keda-central-prd - - contour-internal-0-central-prd - - contour-internal-0-central-prd-intra - - contour-external-central-prd - - external-secrets-central-prd - - flagger-central-prd - - victoriametrics - - kube-system - - monitoring - - telegraf-operator - - loadtester - - opentelemetry - - fluentd - - kube-events - - observability - - contour-cert-checker-ns - - aurva-dataplane - validate: - message: "Deletion of critical namespaces is not allowed." - deny: - conditions: - any: - - key: "{{request.operation}}" - operator: Equals - value: DELETE diff --git a/helm-overrides/gke-central-prd-ase1a/kyverno/policies/restrict-replicas.yaml b/helm-overrides/gke-central-prd-ase1a/kyverno/policies/restrict-replicas.yaml deleted file mode 100644 index d456985..0000000 --- a/helm-overrides/gke-central-prd-ase1a/kyverno/policies/restrict-replicas.yaml +++ /dev/null @@ -1,29 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: restrict-minimum-replicas-selected-ns -spec: - validationFailureAction: Enforce - background: false - rules: - - name: min-replicas-deployments-statefulsets - match: - resources: - kinds: - - Deployment - - StatefulSet - - Deployment/scale - - StatefulSet/scale - names: - - coredns - namespaces: - - kube-system - operations: - - CREATE - - UPDATE - validate: - message: "Deployments and StatefulSets must have at least 5 replicas in this namespace." - anyPattern: - # Case 1: replicas explicitly set and >= 5 - - spec: - replicas: ">=5" diff --git a/helm-overrides/gke-central-prd-ase1a/loadtester/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/loadtester/custom-values.yaml deleted file mode 100644 index ad309f1..0000000 --- a/helm-overrides/gke-central-prd-ase1a/loadtester/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -replicaCount: 1 - -Namespace: loadtester - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/dev/devops/flagger-loadtester - tag: 0.30.0 - pullPolicy: IfNotPresent - pullSecret: - -podLabels: - bu: central - team: central-shared - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - -podPriorityClassName: "" - -logLevel: info -cmd: - timeout: 1h - namespaceRegexp: "" - -nameOverride: "loadtester" -fullnameOverride: "" - -env: [] - -service: - type: ClusterIP - port: 80 - -resources: - requests: - cpu: 10m - memory: 64Mi - -volumes: [] -volumeMounts: [] - -nodeSelector: - cloud.google.com/compute-class: central-devops - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - -affinity: {} - -ingress: - slowStart: - enabled: false - enabled: true - ingressClassName: contour-internal-1 - servicePort: http - hosts: - - host: central-flagger-loadtester.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / -createContourGateway: true -contourResponseTimeout: false -rbac: - # rbac.create: `true` if rbac resources should be created - create: true - # rbac.scope: `cluster` to create cluster-scope rbac resources (ClusterRole/ClusterRoleBinding) - # otherwise, namespace-scope rbac resources will be created (Role/RoleBinding) - scope: - # rbac.rules: array of rules to apply to the role. example: - # rules: - # - apiGroups: [""] - # resources: ["pods"] - # verbs: ["list", "get"] - rules: [] - -# name of an existing service account to use - if not creating rbac resources -serviceAccountName: "" - -# App Mesh virtual node settings (to be used for AppMesh v1beta1) -meshName: "" -#backends: -# - app1.namespace -# - app2.namespace - -# App Mesh virtual node settings (to be used for AppMesh v1beta2) -appmesh: - enabled: false - backends: - - podinfo - - podinfo-canary - -#Istio virtual service and gatway settings. TLS secrets should be in namespace before enbaled it. ( secret format loadtester.fullname ) -istio: - enabled: false - host: flagger-loadtester.flagger - gateway: - enabled: false - tls: - enabled: false - httpsRedirect: false - -# when enabled, it will add a security context for the loadtester pod -securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 100 - runAsGroup: 101 - -podDisruptionBudget: - enabled: false - minAvailable: 1 diff --git a/helm-overrides/gke-central-prd-ase1a/opentelemetry-claude-metrics/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/opentelemetry-claude-metrics/custom-values.yaml deleted file mode 100644 index a98135c..0000000 --- a/helm-overrides/gke-central-prd-ase1a/opentelemetry-claude-metrics/custom-values.yaml +++ /dev/null @@ -1,338 +0,0 @@ -nameOverride: "" -fullnameOverride: "opentelemetry-claude-metrics" - -additionalLabels: - bu: "central" - team: "sre" - service: "opentelemetry-claude-metrics" - env: "prd" - priority: "p1" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: true - key: meesho/prd/cntr/xcntr/otel-claude-metrics - secretStoreRef: - name: vault-backend - -mode: "deployment" - -namespaceOverride: "" - -presets: - logsCollection: - enabled: false - hostMetrics: - enabled: false - kubernetesAttributes: - enabled: false - kubeletMetrics: - enabled: false - kubernetesEvents: - enabled: false - clusterMetrics: - enabled: false - -configMap: - create: true - -config: - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - auth: - authenticator: bearertokenauth - http: - endpoint: ${env:MY_POD_IP}:4318 - auth: - authenticator: bearertokenauth - - processors: - batch: - send_batch_size: 1024 - send_batch_max_size: 2048 - timeout: 10s - memory_limiter: - check_interval: 1s - limit_percentage: 85 - spike_limit_percentage: 20 - - # --- PII redaction for Cowork event bodies (prompt text) -------------- - # Cowork events carry raw prompt text. This drops it BEFORE the events - # reach ClickHouse, while keeping cost/token/tool fields for stats. - # error_mode: ignore => a wrong key is a silent no-op (won't crash the - # pipeline), which ALSO means: do NOT assume prompts are redacted until - # you confirm the real key/location from the debug output. - # - If prompt text is a log ATTRIBUTE -> delete_key(attributes, "") - # - If it is in the log BODY -> set(body, "") (uncomment below) - # Confirm the exact key ("prompt", "prompt.text", …) from step-1 debug logs. - transform/redact: - error_mode: ignore - log_statements: - - context: log - statements: - # - delete_key(attributes, "prompt") # commented: allow prompt text through for observability - # - delete_key(attributes, "prompt.text") # commented: allow prompt text through for observability - # - set(body, "") where attributes["event.name"] == "user_prompt" - - # Pin the HELP (description) string for claude_code.lines_of_code.count so - # a mixed fleet (CLI < 2.1.172 emits the old description; >= 2.1.172 emits - # the new one) doesn't trigger the prometheus exporter's "N error(s) - # occurred for claude_observability_claude_code_lines_of_code_count_total" - # HELP-collision error. If the same pattern shows up on other metrics - # (token.usage, cost.usage, ...), add another `where name == "..."` line - # here rather than defining a second processor. - transform/claude_metrics_help_fix: - error_mode: ignore - metric_statements: - - context: metric - statements: - - set(description, "Count of lines of code modified, with the 'type' attribute indicating whether lines were added or removed and the 'model' attribute indicating which model made the change") where name == "claude_code.lines_of_code.count" - - exporters: - # Claude Code metrics path — unchanged. - prometheus: - endpoint: "0.0.0.0:8889" - namespace: claude_observability - resource_to_telemetry_conversion: - enabled: true - metric_expiration: 5m - - # Cowork events land here as raw OTel logs. - # NOTE: verify DSN scheme (tcp:// vs clickhouse://) and the `ttl` field - # name against the clickhouseexporter README for image tag 0.111.0 — - # both changed across releases and are the most likely mismatch. - clickhouse: - endpoint: http://claude-observability.prd.meesho.int:8080?dial_timeout=10s&compress=lz4 # ClickHouse HTTP interface (DNS — VM IP is not stable) - database: otel - username: claude - password: ${CLICKHOUSE_PASSWORD} # must be injected via the vault secret (see note) - logs_table_name: otel_logs - ttl: 720h # 30d retention, enforced by ClickHouse - create_schema: true # needs CREATE priv on the writer role; else set false + pre-create table - async_insert: true - timeout: 10s - sending_queue: - queue_size: 5000 - retry_on_failure: - enabled: true - - # TEMPORARY: prints full event bodies (incl. prompt text) to stdout. - # Keep for first-deploy validation only, then remove from the logs pipeline. - debug: - verbosity: detailed - - extensions: - health_check: - path: /health - bearertokenauth: - token: ${AUTH_TOKEN} - - service: - telemetry: - metrics: - level: normal - address: ${env:MY_POD_IP}:8888 - extensions: - - health_check - - bearertokenauth - pipelines: - traces: null - # Cowork events arrive as OTLP logs, get redacted, and are stored raw in - # ClickHouse. Aggregation happens at query time in SQL — no connectors. - # `debug` is validation-only; remove it (and the redact caveat aside) - # once you've confirmed events land and the redaction key is correct. - logs: - receivers: - - otlp - processors: - - memory_limiter - - transform/redact - - batch - exporters: - - clickhouse - - debug - # Existing Claude Code metrics only — Cowork no longer feeds this pipeline. - metrics: - receivers: - - otlp - processors: - - memory_limiter - - transform/claude_metrics_help_fix - - batch - exporters: - - prometheus - -image: - repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - tag: "0.111.0" - digest: "" - -imagePullSecrets: [] - -command: - name: otelcol-contrib - extraArgs: [] - -serviceAccount: - create: true - annotations: {} - name: "" - -clusterRole: - create: false - annotations: {} - name: "" - rules: [] - clusterRoleBinding: - annotations: {} - name: "" - -podSecurityContext: {} -securityContext: {} - -nodeSelector: {} - -tolerations: [] - -affinity: {} -topologySpreadConstraints: [] -priorityClassName: "" - -extraEnvs: - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - - name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - -extraEnvsFrom: - - secretRef: - name: opentelemetry-claude-metrics-secret -extraVolumes: [] -extraVolumeMounts: [] - -ports: - otlp: - enabled: true - containerPort: 4317 - servicePort: 4317 - protocol: TCP - appProtocol: grpc - otlp-http: - enabled: true - containerPort: 4318 - servicePort: 4318 - protocol: TCP - prom-exporter: - enabled: true - containerPort: 8889 - servicePort: 8889 - protocol: TCP - jaeger-compact: - enabled: false - jaeger-thrift: - enabled: false - jaeger-grpc: - enabled: false - zipkin: - enabled: false - metrics: - enabled: true - containerPort: 8888 - servicePort: 8888 - protocol: TCP - -resources: - requests: - cpu: 2 - memory: 2Gi - limits: - cpu: 2 - memory: 2Gi - -podAnnotations: - otel.io/path: /metrics - otel.io/port: "8888" - otel.io/scrape: "true" - -podLabels: {} - -hostNetwork: false -dnsPolicy: "ClusterFirstWithHostNet" -dnsConfig: {} - -replicaCount: 2 -revisionHistoryLimit: 10 - -annotations: {} -extraContainers: [] -initContainers: [] -lifecycleHooks: {} - -livenessProbe: - httpGet: - port: 13133 - path: /health - -readinessProbe: - httpGet: - port: 13133 - path: /health - -service: - type: ClusterIP - annotations: - io.cilium/global-service: "true" - cloud.google.com/neg: '{"exposed_ports": {"4318":{"name": "otel-cld-ext-cntr-prd-a"}}}' - -ingress: - enabled: false - -podMonitor: - enabled: false - -serviceMonitor: - enabled: false - -podDisruptionBudget: - enabled: true - maxUnavailable: 1 - -autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 10 - behavior: {} - targetCPUUtilizationPercentage: 70 - targetMemoryUtilizationPercentage: 60 - -rollout: - rollingUpdate: - maxUnavailable: 1 - strategy: RollingUpdate - -prometheusRule: - enabled: false - groups: [] - defaultRules: - enabled: false - extraLabels: {} - -statefulset: - volumeClaimTemplates: [] - podManagementPolicy: "Parallel" - -networkPolicy: - enabled: false - annotations: {} - allowIngressFrom: [] - extraIngressRules: [] - egressRules: [] diff --git a/helm-overrides/gke-central-prd-ase1a/opentelemetry-codex-metrics/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/opentelemetry-codex-metrics/custom-values.yaml deleted file mode 100644 index e4cd24a..0000000 --- a/helm-overrides/gke-central-prd-ase1a/opentelemetry-codex-metrics/custom-values.yaml +++ /dev/null @@ -1,330 +0,0 @@ -nameOverride: "" -fullnameOverride: "opentelemetry-codex-metrics" - -additionalLabels: - bu: "central" - team: "sre" - service: "opentelemetry-codex-metrics" - env: "prd" - priority: "p1" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: true - key: meesho/prd/cntr/xcntr/otel-codex-metrics - secretStoreRef: - name: vault-backend - -mode: "deployment" - -namespaceOverride: "" - -presets: - logsCollection: - enabled: false - hostMetrics: - enabled: false - kubernetesAttributes: - enabled: false - kubeletMetrics: - enabled: false - kubernetesEvents: - enabled: false - clusterMetrics: - enabled: false - -configMap: - create: true - -config: - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - auth: - authenticator: bearertokenauth - http: - endpoint: ${env:MY_POD_IP}:4318 - auth: - authenticator: bearertokenauth - - processors: - batch: - send_batch_size: 1024 - send_batch_max_size: 2048 - timeout: 10s - memory_limiter: - check_interval: 1s - limit_percentage: 85 - spike_limit_percentage: 20 - # codex.tool_result events carry raw shell command arguments and raw - # command output, which can contain secrets. Drop them before export. - # User prompts are already redacted client-side (log_user_prompt=false). - attributes/scrub-sensitive: - actions: - - key: arguments - action: delete - - key: output - action: delete - # Codex uses reasoning_effort on conversation events and - # model_reasoning_effort on completed response events. Normalize both so - # the derived metric exposes one bounded label. - transform/codex-log-attributes: - error_mode: ignore - log_statements: - - context: log - statements: - - set(attributes["codex.reasoning_effort"], attributes["reasoning_effort"]) where attributes["reasoning_effort"] != nil - - set(attributes["codex.reasoning_effort"], attributes["model_reasoning_effort"]) where attributes["codex.reasoning_effort"] == nil and attributes["model_reasoning_effort"] != nil - - connectors: - # Convert selected Codex log dimensions into a Prometheus counter. - # user.email and user.account_id are included intentionally for per-user - # usage attribution (parity with Claude Code telemetry). Do NOT add - # conversation ID, prompt, arguments, or output as labels. - count/codex_logs: - logs: - codex.log.events: - description: Count of Codex OTLP log events by bounded dimensions - conditions: - - 'attributes["event.name"] != nil' - attributes: - - key: event.name - default_value: unknown - - key: event.kind - default_value: unknown - - key: user.account_id - default_value: unknown - - key: user.email - default_value: unknown - - key: codex.reasoning_effort - default_value: unknown - - key: model - default_value: unknown - - key: originator - default_value: unknown - - key: app.version - default_value: unknown - - exporters: - prometheus: - endpoint: "0.0.0.0:8889" - namespace: codex_observability - resource_to_telemetry_conversion: - enabled: true - metric_expiration: 5m - # Phase 1 log sink: registers /v1/logs on the OTLP HTTP receiver and - # surfaces traffic in collector stdout for validation. A follow-up will - # switch this to a persistent store once the backend is finalised. - debug/logs: - verbosity: basic - - extensions: - health_check: - path: /health - bearertokenauth: - token: ${AUTH_TOKEN} - - service: - telemetry: - metrics: - level: normal - address: ${env:MY_POD_IP}:8888 - extensions: - - health_check - - bearertokenauth - pipelines: - traces: null - logs: - receivers: - - otlp - processors: - - memory_limiter - - attributes/scrub-sensitive - - transform/codex-log-attributes - - batch - exporters: - - debug/logs - - count/codex_logs - metrics: - receivers: - - otlp - - count/codex_logs - processors: - - memory_limiter - - batch - exporters: - - prometheus - -image: - repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - tag: "0.111.0" - digest: "" - -imagePullSecrets: [] - -command: - name: otelcol-contrib - extraArgs: [] - -serviceAccount: - create: true - annotations: {} - name: "" - -clusterRole: - create: false - annotations: {} - name: "" - rules: [] - clusterRoleBinding: - annotations: {} - name: "" - -podSecurityContext: {} -securityContext: {} - -nodeSelector: {} - -tolerations: [] - -affinity: {} -topologySpreadConstraints: [] -priorityClassName: "" - -extraEnvs: - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - - name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - -extraEnvsFrom: - - secretRef: - name: opentelemetry-codex-metrics-secret -extraVolumes: [] -extraVolumeMounts: [] - -ports: - otlp: - enabled: true - containerPort: 4317 - servicePort: 4317 - protocol: TCP - appProtocol: grpc - otlp-http: - enabled: true - containerPort: 4318 - servicePort: 4318 - protocol: TCP - prom-exporter: - enabled: true - containerPort: 8889 - servicePort: 8889 - protocol: TCP - jaeger-compact: - enabled: false - jaeger-thrift: - enabled: false - jaeger-grpc: - enabled: false - zipkin: - enabled: false - metrics: - enabled: true - containerPort: 8888 - servicePort: 8888 - protocol: TCP - -resources: - requests: - cpu: 2 - memory: 2Gi - limits: - cpu: 2 - memory: 2Gi - -podAnnotations: - otel.io/path: /metrics - otel.io/port: "8888" - otel.io/scrape: "true" - -podLabels: {} - -hostNetwork: false -dnsPolicy: "ClusterFirstWithHostNet" -dnsConfig: {} - -replicaCount: 2 -revisionHistoryLimit: 10 - -annotations: {} -extraContainers: [] -initContainers: [] -lifecycleHooks: {} - -livenessProbe: - httpGet: - port: 13133 - path: /health - -readinessProbe: - httpGet: - port: 13133 - path: /health - -service: - type: ClusterIP - annotations: - io.cilium/global-service: "true" - cloud.google.com/neg: '{"exposed_ports": {"4318":{"name": "otel-cdx-ext-cntr-prd-a"}}}' - -ingress: - enabled: false - -podMonitor: - enabled: false - -serviceMonitor: - enabled: false - -podDisruptionBudget: - enabled: true - maxUnavailable: 1 - -autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 10 - behavior: {} - targetCPUUtilizationPercentage: 70 - targetMemoryUtilizationPercentage: 60 - -rollout: - rollingUpdate: - maxUnavailable: 1 - strategy: RollingUpdate - -prometheusRule: - enabled: false - groups: [] - defaultRules: - enabled: false - extraLabels: {} - -statefulset: - volumeClaimTemplates: [] - podManagementPolicy: "Parallel" - -networkPolicy: - enabled: false - annotations: {} - allowIngressFrom: [] - extraIngressRules: [] - egressRules: [] diff --git a/helm-overrides/gke-central-prd-ase1a/opentelemetry-coralogix/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/opentelemetry-coralogix/custom-values.yaml deleted file mode 100644 index 35137df..0000000 --- a/helm-overrides/gke-central-prd-ase1a/opentelemetry-coralogix/custom-values.yaml +++ /dev/null @@ -1,277 +0,0 @@ -global: - domain: "coralogixsg.com" - defaultApplicationName: "default" - defaultSubsystemName: "nodes" - - # Old endpoint based configuration, - # please use domain instead. - traces: - endpoint: "" - metrics: - endpoint: "" - logs: - endpoint: "" - -# set distribution to openshift for openshift clusters -distribution: "" - -# Opentelemetry collector configuration -dedicatedValue: false -schedulerName: default-scheduler - -labels: - bu: "central" - team: "sre-shared" - service: "opentelemetry-central-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: false - key: prd/admin/coralogix-keys - secretStoreRef: - name: vault-backend - -image: - # If you want to use the core image `otel/opentelemetry-collector`, you also need to change `command.name` value to `otelcol`. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - # repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - tag: "0.87.0" - # When digest is set to a non-empty value, images will be pulled by digest (regardless of tag value). - digest: "" -mode: daemonset -rollout: - rollingUpdate: - # When 'mode: daemonset', maxSurge cannot be used when hostPort is set for any of the ports - # maxSurge: 25% - maxUnavailable: 10 - strategy: RollingUpdate -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" -hostNetwork: true -dnsPolicy: "ClusterFirstWithHostNet" -fullnameOverride: opentelemetry-central-prd - -presets: - logsCollection: - enabled: false - storeCheckpoints: false - maxRecombineLogSize: 1048576 - extraFilelogOperators: [] -# - type: recombine -# combine_field: body -# source_identifier: attributes["log.file.path"] -# is_first_entry: body matches "^(YOUR-LOGS-REGEX)" - kubernetesAttributes: - enabled: false - hostMetrics: - enabled: false - kubeletMetrics: - enabled: false - -extraEnvs: -- name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" -- name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName -config: - extensions: {} - # zpages: - # endpoint: localhost:55679 - # pprof: - # endpoint: localhost:1777 - exporters: - loadbalancing: - protocol: - otlp: - timeout: 4s - tls: - insecure: true - sending_queue: - enabled: true - queue_size: 50000 - num_consumers: 500 - resolver: - dns: - hostname: "opentelemetry-admin-prd.opentelemetry.svc.clusterset.local" - otlp: - endpoint: opentelemetry-deployment-central-prd.opentelemetry.svc.cluster.local:4317 - tls: - insecure: true - keepalive: - timeout: 2s - # coralogix: - # timeout: "30s" - # private_key: "${CORALOGIX_PRIVATE_KEY}" - # domain: "{{.Values.global.domain}}" - # traces: - # endpoint: "{{ .Values.global.traces.endpoint }}" - # metrics: - # endpoint: "{{ .Values.global.metrics.endpoint }}" - # logs: - # endpoint: "{{ .Values.global.logs.endpoint }}" - # application_name_attributes: - # - "k8s.namespace.name" - # - "service.namespace" - # subsystem_name_attributes: - # - "k8s.deployment.name" - # - "k8s.statefulset.name" - # - "k8s.daemonset.name" - # - "k8s.cronjob.name" - # - "k8s.job.name" - # - "k8s.container.name" - # - "k8s.node.name" - # - "service.name" - # application_name: "{{.Values.global.defaultApplicationName }}" - # subsystem_name: "{{.Values.global.defaultSubsystemName }}" - processors: - k8sattributes: - filter: - node_from_env_var: KUBE_NODE_NAME - extract: - metadata: - - "k8s.namespace.name" - - "k8s.deployment.name" - - "k8s.statefulset.name" - - "k8s.daemonset.name" - - "k8s.cronjob.name" - - "k8s.job.name" - - "k8s.pod.name" - - "k8s.node.name" - memory_limiter: null # Will get the k8s resource limits - groupbytrace: - wait_duration: 1s - groupbyattrs: - keys: - - host.name - resourcedetection/env: - detectors: ["system","env"] - timeout: 5s - override: false - # spanmetrics: - # metrics_exporter: coralogix - # dimensions: - # - name: "k8s.deployment.name" - # - name: "k8s.statefulset.name" - # - name: "k8s.daemonset.name" - # - name: "k8s.cronjob.name" - # - name: "k8s.job.name" - # - name: "k8s.container.name" - # - name: "k8s.node.name" - # - name: "k8s.namespace.name" - receivers: - otlp: - protocols: - grpc: - endpoint: ${MY_POD_IP}:4317 - http: - endpoint: ${MY_POD_IP}:4318 - zipkin: - endpoint: ${MY_POD_IP}:9411 - jaeger: - protocols: - grpc: - endpoint: ${MY_POD_IP}:14250 - thrift_http: - endpoint: ${MY_POD_IP}:14268 - thrift_compact: - endpoint: ${MY_POD_IP}:6831 - thrift_binary: - endpoint: ${MY_POD_IP}:6832 - prometheus: - config: - scrape_configs: - - job_name: opentelemetry-collector - scrape_interval: 30s - static_configs: - - targets: - - ${MY_POD_IP}:8888 - service: - extensions: - # - zpages - # - pprof - - health_check -# - memory_ballast - telemetry: - metrics: - address: ${env:MY_POD_IP}:8888 - pipelines: - traces: - exporters: - - loadbalancing - processors: - - memory_limiter - - batch - receivers: - - otlp - metrics: null - logs: null -tolerations: - - operator: Exists - -resources: - requests: - cpu: 350m - memory: 350Mi - limits: - cpu: 1 - memory: 1G - -ports: - jaeger-binary: - enabled: true - containerPort: 6832 - servicePort: 6832 - hostPort: 6832 - protocol: TCP - # In order to enable podMonitor, following part must be enabled in order to expose the required port: - # metrics: - # enabled: true - -# podMonitor: -# enabled: true - -# prometheusRule: -# enabled: true -# defaultRules: -# enabled: true - -# Annotations to be added to pod -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external - - key: node_pool - operator: NotIn - values: - - np-central-default-prd-ase1 \ No newline at end of file diff --git a/helm-overrides/gke-central-prd-ase1a/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index 12da8c5..0000000 --- a/helm-overrides/gke-central-prd-ase1a/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,119 +0,0 @@ -fullnameOverride: opentelemetry-central-prd - -mode: daemonset - -priorityClassName: "system-node-critical" - -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmstorage - - megaduo - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external - - key: node_pool - operator: NotIn - values: - - np-central-default-prd-ase1 - -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 100m - memory: 124Mi - limits: - cpu: 200m - memory: 256Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "central" - team: "sre" - service: "opentelemetry-central-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/gke-central-prd-ase1a/prometheus-node-exporter/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index 579c955..0000000 --- a/helm-overrides/gke-central-prd-ase1a/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,497 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-admin-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - - -# Extra labels to be added to node exporter pods -podLabels: - bu: "central" - team: "central-sre" - service: "node-exporter-central-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/gke-central-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index 65f336f..0000000 --- a/helm-overrides/gke-central-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-central-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-central-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "central" - team: "central-sre" - service: "stackdriver-exporter-central-prd" - env: "prd" - priority: "p0" - type: "exporter" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-central-prd-0622" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,compute.googleapis.com/instance,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime,loadbalancing.googleapis.com/https' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: - - 'loadbalancing.googleapis.com/https:resource.labels.url_map_name=one_of("ext-lb-prd-cntr-xcntr-edge-guard-url-map","ext-lb-prd-cntr-xcntr-edge-guard-https-redirect","int-lb-prd-cntr-xcntr-edge-guard-url-map")' - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-cntr-cnsre-stackdriver-prd@meesho-central-prd-0622.iam.gserviceaccount.com - } - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/gke-central-prd-ase1a/telegraf-operator/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/telegraf-operator/custom-values.yaml deleted file mode 100644 index b207c77..0000000 --- a/helm-overrides/gke-central-prd-ase1a/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,150 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "central" - team: "central-sre" - service: "telegraf-operator-central-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false diff --git a/helm-overrides/gke-central-prd-ase1a/temporal/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/temporal/custom-values.yaml deleted file mode 100644 index 601430f..0000000 --- a/helm-overrides/gke-central-prd-ase1a/temporal/custom-values.yaml +++ /dev/null @@ -1,157 +0,0 @@ -fullnameOverride: "prd-central-shared-temporal" - -externalSecret: - enabled: true - path: prd/cntr/devop/shared-temporal - refreshInterval: "1m" - secretStoreRef: - kind: ClusterSecretStore - name: vault-backend - -server: - nodeSelector: - cloud.google.com/compute-class: central-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: central-devops - resources: - requests: - cpu: 500m - memory: 1024Mi - limits: - cpu: 500m - memory: 1024Mi - config: - namespaces: - create: true - namespace: - - name: default - retention: 3d - - name: cicd-service - retention: 3d - - name: central-rollout-service - retention: 3d - persistence: - default: - driver: "sql" - sql: - driver: "mysql8" - host: 10.147.2.79 - port: 3306 - database: temporal - user: temporal-user - existingSecret: prd-central-shared-temporal-secret - maxConns: 20 - maxIdleConns: 20 - maxConnLifetime: "1h" - visibility: - driver: "sql" - visibilityStore: es-visibility - sql: - driver: "mysql8" - host: 10.147.2.79 - port: 3306 - database: temporal_visibility - user: temporal-user - existingSecret: prd-central-shared-temporal-secret - maxConns: 20 - maxIdleConns: 20 - maxConnLifetime: "1h" - datastores: - es-visibility: # Define the Elasticsearch datastore connection information under the `es-visibility` key - elasticsearch: - version: "v7" - url: - scheme: "http" - host: "elasticsearch-master-headless:9200" - indices: - visibility: temporal_visibility_v1_dev - -admintools: - nodeSelector: - cloud.google.com/compute-class: central-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: central-devops - resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 100m - memory: 128Mi - -web: - image: - repository: temporalio/ui - tag: 2.39.0 - pullPolicy: IfNotPresent - ingress: - enabled: true - className: contour-internal-1 - annotations: {} - hosts: - - "temporal.prd.meesho.int" - tls: [] - nodeSelector: - cloud.google.com/compute-class: central-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: central-devops - resources: - requests: - cpu: 1000m - memory: 1024Mi - limits: - cpu: 1000m - memory: 1024Mi - -elasticsearch: - enabled: true - replicas: 3 - persistence: - enabled: true - volumeClaimTemplate: - accessModes: ["ReadWriteOnce"] - storageClassName: premium-rwo - resources: - requests: - storage: 50Gi - imageTag: 7.17.3 - host: elasticsearch-master-headless - scheme: http - port: 9200 - version: "v7" - logLevel: "error" - username: "" - password: "" - visibilityIndex: "temporal_visibility_v1_dev" - nodeSelector: - cloud.google.com/compute-class: central-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: central-devops - resources: - requests: - cpu: 1 - memory: 6Gi - limits: - cpu: 1 - memory: 6Gi - -prometheus: - enabled: false -grafana: - enabled: false -cassandra: - enabled: false -mysql: - enabled: true \ No newline at end of file diff --git a/helm-overrides/gke-central-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml deleted file mode 100644 index f4b33a1..0000000 --- a/helm-overrides/gke-central-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml +++ /dev/null @@ -1,296 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-central-a-prd-dr -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-central-a-prd-dr-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent-dr - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-cntr-cnsre-vmagent-prd@meesho-central-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - - https://vminsert-infra-prd-dr.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "central" - team: "central-sre" - service: "vmagent-central-a-prd" - env: "prd" - priority: "p0" - type: "vmagent-dr" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-central-a-prd-dr"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-central-a-prd-dr.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - cloud.google.com/compute-class: "vmagent-dr" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/gke-central-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml deleted file mode 100644 index 22b6e87..0000000 --- a/helm-overrides/gke-central-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml +++ /dev/null @@ -1,272 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 1 - -fullnameOverride: vmagent-central-a-prd-fb -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-central-a-prd-fb-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - -# Remove topology spread constraints -topologySpreadConstraints: [] - -# Modify deployment strategy -deployment: - enabled: true - -# Remove HPA -horizontalPodAutoscaler: - enabled: false - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-infr-sre-vmagent-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -# Remove pod disruption budget -podDisruptionBudget: - enabled: false - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-central.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vmsinglenode-infra-prd.victoriametrics.svc.clusterset.local:8428/api/v1/write - - https://vmsinglenode-infra-prd.meeshogcp.in/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "central-fb" - team: "sre" - service: "vmagent-central-a-prd-fb" - env: "prd" - priority: "p0" - type: "vmagent-fb" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-central-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-central-a-prd-fb.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -# Modify resources -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - requests: - cpu: 2 - memory: 2Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/gke-central-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml deleted file mode 100644 index ee74c8b..0000000 --- a/helm-overrides/gke-central-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert-dr - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-central-a-prd-dr - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/central/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-central-a-prd-dr.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-central-a-prd-dr" - env: "prd" - priority: "p0" - type: "vmalert-dr" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmagent-dr" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-central-prd-ase1a/victoria-metrics-alert-secured-stateful/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/victoria-metrics-alert-secured-stateful/custom-values.yaml deleted file mode 100644 index 84f0a46..0000000 --- a/helm-overrides/gke-central-prd-ase1a/victoria-metrics-alert-secured-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-stateful-secured-central-a-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-central-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-central-a-prd.victoriametrics.svc.cluster.local:8480/insert/200/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-central-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/sensitive/central/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: false - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-central-a-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "central" - team: "sre" - service: "vmalert-stateful-secured-central-a-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-secured-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-central-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml deleted file mode 100644 index b319296..0000000 --- a/helm-overrides/gke-central-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-central-a-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-central-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-central-a-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-central-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/central/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-central-a-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "central" - team: "sre" - service: "vmalert-central-a-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-central-prd-ase1a/victoriametrics-agent/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index ff31309..0000000 --- a/helm-overrides/gke-central-prd-ase1a/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,484 +0,0 @@ -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "central-a-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -# -- Replica count -replicaCount: 2 - -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - -# -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables -deployment: - spec: - minReadySeconds: 1200 - progressDeadlineSeconds: 1800 - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - -image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - # -- Image tag, set to `Chart.AppVersion` by default - tag: v1.133.0 # rewrites Chart.AppVersion - # -- Variant of the image to use. - # e.g. enterprise, scratch - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - -# -- Image pull secrets -imagePullSecrets: [] - -# -- Add additional DNS entries to pods hosts file. Check [official documentation](https://kubernetes.io/docs/tasks/network/customize-hosts-file-for-pods/) -hostAliases: [] -# - ip: 192.168.1.1 -# hostNames: -# - test.example.com -# - another.example.net - -# -- Override chart name -nameOverride: "" - -# -- Override resources fullname -fullnameOverride: "vm-agent-central-a-prd" - -# -- Container working directory -containerWorkingDir: "/" - -rbac: - # -- Enables Role/RoleBinding creation - create: true - - # -- Role/RoleBinding annotations - annotations: {} - - # -- Role/RoleBinding labels - extraLabels: {} - - # -- If true and `rbac.enabled`, will deploy a Role/RoleBinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - - # -- additional rules for a role - extraRules: [] - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - # -- Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-cntr-sre-vmagnt-prd-mds@meesho-central-prd-0622.iam.gserviceaccount.com - } - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # -- mount API token to pod directly - automountToken: true - -# -- See `kubectl explain poddisruptionbudget.spec` for more or check [official documentation](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# -- Generates `remoteWrite.*` flags and config maps with value content for values, that are of type list of map. -# Each item should contain `url` param to pass validation. -remoteWrite: - # - url: http://vm-insert-central-prd.victoriametrics.svc.clusterset.local:8480/insert/multitenant/prometheus/api/v1/write - # disableOnDiskQueue: false - # dropSamplesOnOverload: false - - url: http://vm-insert-central-a-prd.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write - disableOnDiskQueue: false - dropSamplesOnOverload: false - - url: http://prd-census-server.prd-census-server.svc.cluster.local/api/v1/write - disableOnDiskQueue: true - dropSamplesOnOverload: true - -# -- VMAgent extra command line arguments -extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8429 - promscrape.config.strictParse: false # default is true - promscrape.maxScrapeSize: 1000000000 # default is 167772160 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - # promscrape.suppressScrapeErrors: true # default is false - - # Uncomment and specify the port if you want to support any of the protocols: - # https://docs.victoriametrics.com/victoriametrics/vmagent/#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for more details. -env: - - name: GOGC - value: "200" - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth-secret - # key: password - -# -- Specify alternative source for env variables -envFrom: - [] - #- configMapRef: - # name: special-config - -# -- Extra labels for Deployment and Statefulset -extraLabels: - bu: "central" - team: "central-sre" - service: "vm-agent-central-a-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra labels for Pods only -podLabels: - bu: "central" - team: "central-sre" - service: "vm-agent-central-a-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra selector labels common for pod and service -selectorLabels: {} - -# -- Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# -- Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# -- Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -# -- Extra containers to run in a pod with vmagent -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -# -- Init containers for vmagent -initContainers: - [] - # - name: example - # image: example-image - -# -- Security context to be added to pod -podSecurityContext: - enabled: true - # fsGroup: 2000 - -# -- Security context to be added to pod's containers -securityContext: - enabled: true - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - # -- Enable agent service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - extraLabels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) for details - externalIPs: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8429 - # -- Target port - targetPort: http - # nodePort: 30000 - # -- Service type - type: ClusterIP - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - -ingress: - # -- Enable deployment of ingress for agent - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-agent-central-a-prd.meeshogcp.in - path: - - / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - - # -- Ingress controller class name - ingressClassName: "nginx-internal" - - # -- Ingress path type - pathType: Prefix - -# -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) -resources: - requests: - cpu: 27 - memory: 50Gi - -# -- Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) -nodeSelector: - cloud.google.com/compute-class: "vmagent-mds" - -# -- Node tolerations for server scheduling to nodes with taints. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-mds" - effect: "NoSchedule" - -# -- Pod topologySpreadConstraints -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# -- Pod affinity -affinity: {} - -# -- VMAgent [scraping configuration](https://docs.victoriametrics.com/victoriametrics/vmagent/#how-to-collect-metrics-in-prometheus-format) -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-central-a-prd" - -# -- Priority class to be assigned to the pod(s) -priorityClassName: "" - -# -- Enable the host network -hostNetwork: false - -serviceMonitor: - # -- Enable deployment of Service Monitor for server component. This is Prometheus operator object - enabled: false - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Service Monitor relabelings - relabelings: [] - # -- Basic auth params for Service Monitor - basicAuth: {} - # -- Service Monitor metricRelabelings - metricRelabelings: [] - # -- Service Monitor targetPort - targetPort: http - # interval: 15s - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - -# -- Empty dir configuration for a case, when persistence is disabled -emptyDir: {} - -persistentVolume: - # -- Create/use Persistent Volume Claim for server component. Empty dir if false - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Size of the volume. Should be calculated based on the logs you send and retention policy you set. - size: 10Gi - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume additional labels - extraLabels: {} - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Horizontal Pod Autoscaling. -# Note that it is not intended to be used for vmagents which perform scraping. -# In order to scale scraping vmagents check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) -horizontalPodAutoscaling: - # -- Use HPA for vmagent - enabled: false - # -- Maximum replicas for HPA to use to to scale vmagent - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale vmagent - minReplicas: 1 - # -- Metric for HPA to use to scale vmagent - metrics: [] - -# -- VMAgent scrape configuration -config: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -probe: - # -- Readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - # -- Liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - # -- Startup probe - startup: {} - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -allowedMetricsEndpoints: - - /metrics - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-central-prd-ase1a/victoriametrics-insert/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/victoriametrics-insert/custom-values.yaml deleted file mode 100644 index 6b6b6b7..0000000 --- a/helm-overrides/gke-central-prd-ase1a/victoriametrics-insert/custom-values.yaml +++ /dev/null @@ -1,411 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: false - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- IDs of vmstorage nodes to exclude from writing - excludeStorageIDs: [] - # -- Override default `app` label name - name: vminsert - # -- VMInsert strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - # override Chart.AppVersion - tag: v1.107.0-cluster - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMInsert http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-central-a-prd - # -- Extra command line arguments for vminsert component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8480 - maxLabelsPerTimeseries: 55 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-central-a-prd-0.vm-storage-central-a-prd.victoriametrics.svc:8400" - - "vm-storage-central-a-prd-1.vm-storage-central-a-prd.victoriametrics.svc:8400" - - "vm-storage-central-a-prd-2.vm-storage-central-a-prd.victoriametrics.svc:8400" - - "vm-storage-central-a-prd-3.vm-storage-central-a-prd.victoriametrics.svc:8400" - - "vm-storage-central-a-prd-4.vm-storage-central-a-prd.victoriametrics.svc:8400" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "central" - team: "central-sre" - service: "vm-insert-central-a-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Pod's additional labels - podLabels: - bu: "central" - team: "central-sre" - service: "vm-insert-central-a-prd" - env: "prd" - priority: "p0" - type: "vminsert" - terminationGracePeriodSeconds: 30 - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - - # -- Readiness & Liveness probes - probe: - # -- VMInsert readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert startup probe - startup: {} - - # -- Relabel configuration - relabel: - enabled: false - config: [] - # -- Use existing configmap if specified - # otherwise .config values will be used. Relabel config **should** reside under `relabel.yml` key - configMap: "" - - # Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 15 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 4 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vminsert - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vminsert - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - cloud.google.com/compute-class: "vminsert-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 6 - memory: 10Gi - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - service: - # -- Create VMInsert service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here]( https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) for details - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-insert-central-a-prd.meeshogcp.in - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - - # -- Ingress controller class name - ingressClassName: nginx-internal - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for insert component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/insert" .Values.vminsert }}' - - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vminsert component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-central-prd-ase1a/victoriametrics-select/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/victoriametrics-select/custom-values.yaml deleted file mode 100644 index 911d61c..0000000 --- a/helm-overrides/gke-central-prd-ase1a/victoriametrics-select/custom-values.yaml +++ /dev/null @@ -1,449 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - splitService: true - externalService: - enabled: true - name: vmselect-central-a-prd-proxy - # -- Override default `app` label name - - clusternativeService: - enabled: false - - name: "" - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # override Chart.AppVersion - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMSelect http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-central-a-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - # -- Extra command line arguments for vmselect component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8481 - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 180s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 96 - clusternative.maxConcurrentRequests: 256 - storageNode: - - "vm-storage-central-a-prd-0.vm-storage-central-a-prd.victoriametrics.svc:8401" - - "vm-storage-central-a-prd-1.vm-storage-central-a-prd.victoriametrics.svc:8401" - - "vm-storage-central-a-prd-2.vm-storage-central-a-prd.victoriametrics.svc:8401" - - "vm-storage-central-a-prd-3.vm-storage-central-a-prd.victoriametrics.svc:8401" - - "vm-storage-central-a-prd-4.vm-storage-central-a-prd.victoriametrics.svc:8401" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "central" - team: "central-sre" - service: "vm-select-central-a-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # -- Pod's additional labels - podLabels: - bu: "central" - team: "central-sre" - service: "vm-select-central-a-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Readiness & Liveness probes - probe: - # -- VMSelect readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 5 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vmselect - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vmselect - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 18 - memory: 30Gi - - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Create VMSelect service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: false - - # -- Ingress annotations - annotations: {} - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-select-central.prd.meesho.int - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # -- Ingress controller class name - ingressClassName: contour-internal-0 - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for select component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/select" .Values.vmselect }}' - - # -- vmselect mode: deployment, daemonSet - mode: deployment - - # -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables - deployment: - spec: - # -- VMSelect strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - - # -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables - statefulSet: - enabled: false - spec: - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume extra labels - extraLabels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmselect component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: false - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - diff --git a/helm-overrides/gke-central-prd-ase1a/victoriametrics-storage/custom-values.yaml b/helm-overrides/gke-central-prd-ase1a/victoriametrics-storage/custom-values.yaml deleted file mode 100644 index dd30454..0000000 --- a/helm-overrides/gke-central-prd-ase1a/victoriametrics-storage/custom-values.yaml +++ /dev/null @@ -1,363 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - enabled: false - -vminsert: - enabled: false - -vmauth: - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- Override default `app` label name - name: vmstorage - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Variant of the image to use. e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMStorage http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-central-a-prd - - # -- Additional environment variables (ex.: secret tokens, flags). Check https://docs.victoriametrics.com/victoriametrics/#environment-variables for details - env: [] - # -- Specify alternative source for env variables - envFrom: [] - # -- Data retention period. Possible units character: h(ours), d(ays), w(eeks), y(ears), if no unit character specified - month. The minimum retention period is 24h. - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - httpListenAddr: :8482 - loggerTimezone: "Asia/Kolkata" - dedup.minScrapeInterval: 60s - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmstorage-n4d" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - cloud.google.com/compute-class: "vmstorage-n4d" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - # -- Use an alternate scheduler, e.g. "stork". Check https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ for details - # schedulerName: - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Override Persistent Volume Claim name - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are https://kubernetes.io/docs/concepts/storage/persistent-volumes/ - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume extra labels - extraLabels: {} - # -- Storage class name. Will be empty if not set - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume - size: 1200Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "central" - team: "central-sre" - service: "vm-storage-central-a-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - # -- Pod's additional labels - podLabels: - bu: "central" - team: "central-sre" - service: "vm-storage-central-a-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Details are https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 25 - memory: 225Gi - - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: - enabled: false - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - podSecurityContext: - enabled: false - service: - enabled: true - # -- Service annotations - annotations: {} - # -- Service ClusterIP - clusterIP: None - # -- Service type - type: ClusterIP - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmselect - vmselectPort: 8401 - # -- Extra service ports - extraPorts: [] - # -- Health check node port for a service. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilyPolicy: "" - # -- List of service IP families. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilies: [] - # -- Traffic Distribution. Check https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution - trafficDistribution: "" - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - minReadySeconds: 5 - # -- Readiness probes - probe: - # -- VMStorage readiness probe - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMStorage startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmstorage component - enabled: false - # -- Maximum replicas for HPA to use to to scale the vmstorage component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmstorage component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmstorage component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: - scaleDown: - selectPolicy: Disabled - - vmbackupmanager: - # -- Enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-dataengg-prd-ase1a/README.md b/helm-overrides/gke-dataengg-prd-ase1a/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/gke-dataengg-prd-ase1a/aurva-dataplane/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/aurva-dataplane/custom-values.yaml deleted file mode 100644 index 08b4715..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/aurva-dataplane/custom-values.yaml +++ /dev/null @@ -1,684 +0,0 @@ -## Aurva Data Plane -## Ref: https://github.com/aurva-io/aurva-charts.git - -postgresql: - enabled: true - fullnameOverride: "aurva-dataplane-database" - volumePermissions: - ## @param volumePermissions.enabled Enable init container that changes the owner and group of the persistent volume - ## - enabled: true - global: - storageClass: hyperdisk-balanced - postgresql: - auth: - postgresPassword: "aurva" - database: "controller" - # Add toleration to make sure where this postgres db pod should reside (Applicable for production workloads): For more detail ref: https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/ - primary: - extendedConfiguration: | - max_connections = 300 - - #PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: dataengg-devops - - # -- Select nodes to deploy which matches the following labels - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: dataengg-devops - -# -- Provide a name in place of `aurva` -# namespaceOverride: aurva-dataplane - -########################################################## -# Global Configs -########################################################## -global: - aurva_controller: - enabled: true - aurva_fastdet: - enabled: false - aurva_pii_analyzer: - enabled: true - aurva_ocr: - enabled: true - aurva_collector: - enabled: true - - deploymentAnnotations: {} - - priorityClassName: "" - -########################################################## -# Aurva Controller -########################################################## -aurva_controller: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "dataengg" - team: "dataengg-devops" - service: "aurva-dataengg-a-prd" - env: "prd" - priority: "p0" - type: "aurva_controller" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 10 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - image: - # Image of the app container - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-controller/aurva-controller - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - - type: secret - name: aurva-controller-secrets - # - type: configmap - # name: proxy-datasource-config - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 2 - requests: - memory: 1Gi - cpu: 1 - - aurvaFastdet: - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-fastdet/aurva-fastdet - tag: "v2.30.13" - pullPolicy: IfNotPresent - envFrom: [] - env: [] - resources: - limits: - cpu: 0.5 - memory: 512Mi - requests: - cpu: 0.5 - memory: 512Mi - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: dataengg-devops - - # -- Taint tolerations for nodes - ##PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: dataengg-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-controller-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - config: - #variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - COMMAND_URL: "command.aurva-prd.meeshogcp.in:80" - DEPLOYMENT_TYPE: "kubernetes" - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - FLUSHER_WORKER_POOL_SIZE: "1000" - FLUSHER_BATCH_SIZE: "30000" - UNIQUENESS_IDENTIFIER: "gke-dataengg-prd-ase1a" #Recommendation: should be equal to cluster name - PROVIDER_ACCOUNT_ID: "meesho-dataengg-prd-0622" # GCP PROJECT ID (not Number) - REGION: "asia-southeast1" #eg: asia-south1 - ENVIRONMENT: "prod" - OCR_ENABLED: "true" - MONITORING_ENABLED: "false" - HYBRID_ONLY_MODE: "false" - FORCE_TLS: "false" - FASTDET_FLAG : "true" - AADHAAR_ENHANCER: "0" - WORKSPACE_EVENT_TRACKING_ENABLED: "false" - ACCESS_IQ_ENABLED: "true" - GRPC_ENFORCE_ALPN_ENABLED: "false" - ENABLE_FASTDET: "true" - FASTDET_MAX_BATCH_SIZE: "100" - HEARTBEAT_INTERVAL: "5m" - #pii data - PII_BUCKET_NAME: "gcs-infra-devop-aurva-dataengg-prd" - PII_LOG_BUCKET_REGION: "asia-southeast1" - PII_LOG_CRON: "*/5 * * * *" - ENABLE_PII_LOG: "true" - PII_EVIDENCE_UPLOAD_MAX_BLOCKING_TASKS: "50000" - PII_EVIDENCE_UPLOAD_MAX_CONCURRENT_TASKS: "500" - PII_EVIDENCE_MAX_CACHE_WEIGHT: "100" - QUOTA_CLEANUP_CRON: "0 0 * * *" - ENABLE_PII_QUOTA: "true" - MAX_PII_EVIDENCES_PER_KEY_PER_WINDOW: "3" - PII_QUOTA_SYNC_CRON: "*/2 * * * *" - QUOTA_WINDOW_HOURS: "24" - #constants - SKIP_NAMESPACES: "contour-internal-1-dataengg-a-prd,contour-internal-0-dataengg-a-prd-intra,contour-internal-0-dataengg-a-prd,contour-external-dataengg-a-prd" - CLOUD_PROVIDER: "gcp" - LOG_ENV: "production" - RDS_SCANNER_AVAILABILITY : "false" - REDSHIFT_SCANNER_AVAILABILITY : "false" - S3_SCANNER_AVAILABILITY : "false" - DYNAMO_SCANNER_AVAILABILITY: "false" - DOCDB_SCANNER_AVAILABILITY: "false" - OPENSEARCH_SCANNER_AVAILABILITY: "false" - CLOUDSQL_SCANNER_AVAILABILITY: "true" - BIGQUERY_SCANNER_AVAILABILITY: "true" - AWS_SNAPSHOT_SCANNER_AVAILABILITY: "false" - CLOUDSTORAGE_SCANNER_AVAILABILITY: "true" - KEYSPACES_SCANNER_AVAILABILITY: "false" - ALLOYDB_SCANNER_AVAILABILITY: "true" - BIGTABLE_SCANNER_AVAILABILITY: "true" - GCP_BACKUP_AVAILABILITY: "true" - EGRESS_MODE_ONLY: "false" - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - SCAN_UUID_ENABLED: "true" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-controller-sa - # -- Annotations applied to created service account - annotations: - iam.gke.io/gcp-service-account: sa-dataengg-prd-aurva-contr@meesho-dataengg-prd-0622.iam.gserviceaccount.com -# eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 15 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 - - -########################################################## -# Aurva OCR -########################################################## -aurva_ocr: - # -- Additional labels for aurva-controller - additionalLabels: - bu: "dataengg" - team: "dataengg-devops" - service: "aurva-dataengg-a-prd" - env: "prd" - priority: "p0" - type: "aurva_ocr" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 1 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-ocr/aurva-ocr - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-ocr: - type: secret - name: aurva-ocr-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 1 - requests: - memory: 2Gi - cpu: 1 - - # -- Select nodes to deploy which matches the following labels - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: dataengg-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: dataengg-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-ocr-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - OCR_TIME_LIMIT: "1" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "gke-dataengg-prd-ase1a" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-ocr-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # iam.gke.io/gcp-service-account: service-account@gcp.iam.gserviceaccount.com - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva Collector -########################################################## -aurva_collector: - - # -- Additional labels for aurva-analyzer - additionalLabels: - bu: "dataengg" - team: "dataengg-devops" - service: "aurva-dataengg-a-prd" - env: "prd" - priority: "p0" - type: "aurva_collector" - - - # -- Annotations on aurva-analyzer - annotations: {} -# "key": "value" - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-collector/aurva-collector - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-controller: - type: secret - name: aurva-collector-secrets - - resources: - limits: - cpu: 800m - memory: 800Mi - requests: - cpu: 200m - memory: 512Mi - - podSecurityContext: {} - - securityContext: - privileged: true - capabilities: - add: - # For kernel v5.8 and above we don't need CAP_SYS_ADMIN or CAP_SYS_RESOURCE - # we just need CAP_BPF and CAP_PERFMON. This has been tested on our EKS node - # which is on kernel v5.10.x - # When SSL Tracing is required we need CAP_SYS_ADMIN and CAP_SYS_PTRACE - # on top of the previous capabilities - # So finally these are the 4 possible combinations for capabilies - # 1. Newer Kernels without SSL - # - BPF - # - PERFMON - # 2. Newer Kernels with SSL - - SYS_ADMIN - - SYS_PTRACE - # 3. Older Kernels without SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # 4. Older Kernels with SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # - SYS_PTRACE - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - volumes: - - name: debugfs - mountPath: /sys/kernel/debug - hostPath: /sys/kernel/debug - - name: vmlinux - mountPath: /sys/kernel/btf/vmlinux - hostPath: /sys/kernel/btf/vmlinux - - name: procfs - mountPath: /host/proc - hostPath: /proc - - name: bpffs - mountPath: /sys/fs/bpf - hostPath: /sys/fs/bpf - - # -- Taint tolerations for nodes - tolerations: - # - effect: NoSchedule - # key: dedicated - # operator: Equal - # value: megatetra - - operator: Exists - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmagent-mds - - vminsert-mds - - vmselect-mds - - vmstack-startree - - vmstorage-n4d - - contour-external-cc-v1 - - contour-external - - contour-internal-0-cc-v1 - - contour-internal-0 - - contour-internal-1-cc-v1 - - contour-internal-1 - - contour-intra-0-cc-v1 - - contour-intra-1-cc-v1 - - contour-intra-1 - - contour-shared-cc-v1 - - contour-shared - - alloy - - preprod-spot-16 - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-collector-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - # variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "gke-dataengg-prd-ase1a" - DEPLOYMENT_TYPE: "kubernetes" - TRACE_INTERNAL_SVC: "true" - TRACE_INTERNAL_SVC_HTTP: "true" - INTERNAL_SVC_SAMPLE_INTERVAL: "10m" - LOGS_TTL: "1h" - TRACE_HTTP2: "true" - TRACE_SSL: "false" - TRACE_PSQL: "false" - TRACE_SQLSERVER: "false" - TRACE_MYSQL: "false" - TRACE_EGRESS: "true" - TRACE_GO_TLS: "false" - TRACE_ML_SERVICES: "false" - # constants - LOG_ENV: production - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - MONITORING_ENABLED: "false" - ENABLE_INGRESS_INFORMER: "false" - ENABLE_SERVICE_INFORMER: "false" - ENABLE_ISTIO_INFORMER: "false" - EXCLUDED_PII_REGEX_TYPES: "ip_address,us_bank_number,us_driver_license,us_itin,us_passport,us_routing,us_mbi,ssn" - AGGREGATOR_MAX_CONNECTIONS: "1000" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-collector-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva PII Analyzer -########################################################## -aurva_pii_analyzer: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "dataengg" - team: "dataengg-devops" - service: "aurva-dataengg-a-prd" - env: "prd" - priority: "p0" - type: "aurva_pii_analyzer" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 3 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - ##PLACEHOLDER## - nodeSelector: - cloud.google.com/compute-class: dataengg-devops - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-piianalyzer/aurva-piianalyzer - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-pii-analyzer: - type: secret - name: aurva-pii-analyzer-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 4Gi - cpu: 4 - requests: - memory: 2Gi - cpu: 2 - - # nodeSelector: ##PLACEHOLDER## - # cloud.google.com/compute-class: dataengg-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: dataengg-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-pii-analyzer-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - SCHEDULER_TIME: "1" - SUPPORTED_REGION: "US" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "gke-dataengg-prd-ase1a" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-pii-analyzer-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 3 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 diff --git a/helm-overrides/gke-dataengg-prd-ase1a/cert-manager/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/cert-manager/custom-values.yaml deleted file mode 100644 index e81dd1b..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/cert-manager/custom-values.yaml +++ /dev/null @@ -1,129 +0,0 @@ -global: - logLevel: 2 - rbac: - create: true - priorityClassName: "high-priority" -installCRDs: false - -crds: - enabled: true - keep: true - -# Cert-manager Controller -replicaCount: 1 -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-controller - tag: v1.20.2 - pullPolicy: IfNotPresent - -nodeSelector: - cloud.google.com/compute-class: dataengg-devops - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "dataengg-devops" - effect: "NoSchedule" - -resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 500m - memory: 512Mi - -# Webhook Configuration -webhook: - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-webhook - tag: v1.20.2 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: dataengg-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "dataengg-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# CA Injector Configuration -cainjector: - enabled: true - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-cainjector - tag: v1.20.2 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: dataengg-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "dataengg-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# ACME Solver Configuration -acmesolver: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-acmesolver - tag: v1.20.2 - pullPolicy: IfNotPresent - -# Startup API Check -startupapicheck: - enabled: true - timeout: 1m - backoffLimit: 4 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-startupapicheck - tag: v1.20.2 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: dataengg-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "dataengg-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - -# Prometheus Monitoring -prometheus: - enabled: true - servicemonitor: - enabled: false - interval: 60s - scrapeTimeout: 30s - labels: - prometheus: cert-manager diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/16c-104g-dp-dpcon-tco-farm-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/16c-104g-dp-dpcon-tco-farm-cc.yaml deleted file mode 100644 index f308337..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/16c-104g-dp-dpcon-tco-farm-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: 16c-104g-dp-dpcon-tco-farm -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n1-highmem-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 200 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/16c-104g-farm-dp-dpcon-tworker-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/16c-104g-farm-dp-dpcon-tworker-cc.yaml deleted file mode 100644 index 1066ff5..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/16c-104g-farm-dp-dpcon-tworker-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: 16c-104g-farm-dp-dpcon-tworker -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n1-highmem-16 - maxPodsPerNode: 16 - spot: true - storage: - bootDiskSize: 200 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/52c-150g-deng-dpexp-ab-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/52c-150g-deng-dpexp-ab-cc.yaml deleted file mode 100644 index d832930..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/52c-150g-deng-dpexp-ab-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: 52c-150g-deng-dpexp-ab -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-custom-52-153600 - maxPodsPerNode: 24 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/8c-64g-dp-dpnrt-druid-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/8c-64g-dp-dpnrt-druid-cc.yaml deleted file mode 100644 index b9382f3..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/8c-64g-dp-dpnrt-druid-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: 8c-64g-dp-dpnrt-druid -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highmem-8 - maxPodsPerNode: 16 - spot: true - storage: - bootDiskSize: 200 - bootDiskType: pd-ssd - - machineType: c3d-highmem-8 - maxPodsPerNode: 16 - spot: true - storage: - bootDiskSize: 200 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/8c-64g-dp-dpnrt-druid-od-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/8c-64g-dp-dpnrt-druid-od-cc.yaml deleted file mode 100644 index 6cbb032..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/8c-64g-dp-dpnrt-druid-od-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: 8c-64g-dp-dpnrt-druid-od -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highmem-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/alluxio-32c-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/alluxio-32c-cc.yaml deleted file mode 100644 index 00c103d..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/alluxio-32c-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: alluxio-32c -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highmem-32 - maxPodsPerNode: 24 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/alluxio-hmem32-8-lssd-prd-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/alluxio-hmem32-8-lssd-prd-cc.yaml deleted file mode 100644 index 41cd99a..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/alluxio-hmem32-8-lssd-prd-cc.yaml +++ /dev/null @@ -1,23 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: alluxio-hmem32-8-lssd-prd -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highmem-32 - maxPodsPerNode: 24 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - localSSDCount: 8 - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/c3d-hmem-8-sp-dpnrt-druid-int-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/c3d-hmem-8-sp-dpnrt-druid-int-cc.yaml deleted file mode 100644 index 1afa177..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/c3d-hmem-8-sp-dpnrt-druid-int-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: c3d-hmem-8-sp-dpnrt-druid-int -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3d-highmem-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 200 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/compactocta-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/compactocta-cc.yaml deleted file mode 100644 index ca6a37b..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/compactocta-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: compactocta -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highmem-8 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/compacttetra-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/compacttetra-cc.yaml deleted file mode 100644 index 840a3d8..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/compacttetra-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: compacttetra -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-standard-8 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: n2-standard-8 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-external-cc-v1.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-external-cc-v1.yaml deleted file mode 100644 index a79267a..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-external-cc-v1.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-external-cc-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-external-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-external-cc.yaml deleted file mode 100644 index 11a8ae7..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-external-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-external -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c3-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-internal-0-cc-v1.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-internal-0-cc-v1.yaml deleted file mode 100644 index 83f4de4..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-internal-0-cc-v1.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-0-cc-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-internal-0-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-internal-0-cc.yaml deleted file mode 100644 index d1b3004..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-internal-0-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-0 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-internal-1-cc-v1.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-internal-1-cc-v1.yaml deleted file mode 100644 index b05c8f3..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-internal-1-cc-v1.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-1-cc-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-32 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-32 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-internal-1-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-internal-1-cc.yaml deleted file mode 100644 index b9834fb..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-internal-1-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-32 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-32 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-intra-0-cc-v1.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-intra-0-cc-v1.yaml deleted file mode 100644 index 6af153a..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-intra-0-cc-v1.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-0-cc-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-intra-1-cc-v1.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-intra-1-cc-v1.yaml deleted file mode 100644 index 19e8944..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-intra-1-cc-v1.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-1-cc-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-intra-1-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-intra-1-cc.yaml deleted file mode 100644 index b71866e..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-intra-1-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-shared-cc-v1.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-shared-cc-v1.yaml deleted file mode 100644 index efb6c0f..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-shared-cc-v1.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-shared-cc-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-standard-4 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-standard-4 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-shared-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-shared-cc.yaml deleted file mode 100644 index f981b82..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/contour-shared-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-shared -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-standard-4 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp \ No newline at end of file diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dataengg-devops.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dataengg-devops.yaml deleted file mode 100644 index 8deeae3..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dataengg-devops.yaml +++ /dev/null @@ -1,34 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dataengg-devops -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-standard-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 60 - bootDiskType: hyperdisk-balanced - - machineType: c4-standard-4 - spot: false - maxPodsPerNode: 32 - storage: - bootDiskSize: 60 - bootDiskType: hyperdisk-balanced - - machineType: e2-standard-4 - spot: false - maxPodsPerNode: 32 - storage: - bootDiskSize: 60 - bootDiskType: pd-balanced - whenUnsatisfiable: DoNotScaleUp \ No newline at end of file diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dp-airflow-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dp-airflow-cc.yaml deleted file mode 100644 index d3cbc06..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dp-airflow-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dp-airflow -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-standard-16 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 200 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dp-dpnrt-zookeeper-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dp-dpnrt-zookeeper-cc.yaml deleted file mode 100644 index 5ae0f37..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dp-dpnrt-zookeeper-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dp-dpnrt-zookeeper -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-standard-4 - maxPodsPerNode: 22 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-alluxio-n2-hmem-8-c-od.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-alluxio-n2-hmem-8-c-od.yaml deleted file mode 100644 index d69ed96..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-alluxio-n2-hmem-8-c-od.yaml +++ /dev/null @@ -1,25 +0,0 @@ -# Missing on gke-dataengg-prd-ase1a. Referenced by coordinator of: os-trino-mb-ext-bk2 -# Source: nodepool dpcon-alluxio-n2-highmem-8-c-od (n2-highmem-8, on-demand) on k8s-dataengg-prd-ase1. -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dpcon-alluxio-n2-hmem-8-c-od -spec: - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - activeMigration: - optimizeRulePriority: false - priorities: - - machineType: n2-highmem-8 - spot: false - maxPodsPerNode: 24 - storage: - bootDiskType: pd-ssd - bootDiskSize: 100 - priorityDefaults: - location: - zones: - - asia-southeast1-a - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-32-a-co-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-32-a-co-cc.yaml deleted file mode 100644 index 75216e1..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-32-a-co-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dpcon-n2-hmem-32-a-co -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highmem-32 - maxPodsPerNode: 20 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: n4-highmem-32 - maxPodsPerNode: 20 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-32-b-co-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-32-b-co-cc.yaml deleted file mode 100644 index a674e82..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-32-b-co-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dpcon-n2-hmem-32-b-co -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highmem-32 - maxPodsPerNode: 20 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: n4-highmem-32 - maxPodsPerNode: 20 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-32-c-co-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-32-c-co-cc.yaml deleted file mode 100644 index 1b0a802..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-32-c-co-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dpcon-n2-hmem-32-c-co -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highmem-32 - maxPodsPerNode: 20 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: n4-highmem-32 - maxPodsPerNode: 20 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-a-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-a-cc.yaml deleted file mode 100644 index 4b86438..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-a-cc.yaml +++ /dev/null @@ -1,43 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dpcon-n2-hmem-48-a -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: - - asia-southeast1-a - - asia-southeast1-b - - asia-southeast1-c - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highmem-48 - maxPodsPerNode: 28 - spot: true - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: n4-highmem-48 - maxPodsPerNode: 28 - spot: true - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n2-highmem-48 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: n4-highmem-48 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-a-co-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-a-co-cc.yaml deleted file mode 100644 index faed480..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-a-co-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dpcon-n2-hmem-48-a-co -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highmem-48 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: n4-highmem-48 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-b-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-b-cc.yaml deleted file mode 100644 index f92e06a..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-b-cc.yaml +++ /dev/null @@ -1,40 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dpcon-n2-hmem-48-b -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a','asia-southeast1-b','asia-southeast1-c'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highmem-48 - maxPodsPerNode: 28 - spot: true - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: n4-highmem-48 - maxPodsPerNode: 28 - spot: true - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n2-highmem-48 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: n4-highmem-48 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-b-co-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-b-co-cc.yaml deleted file mode 100644 index 6775900..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-b-co-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dpcon-n2-hmem-48-b-co -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a','asia-southeast1-b','asia-southeast1-c'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highmem-48 - maxPodsPerNode: 20 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: n4-highmem-48 - maxPodsPerNode: 20 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-c-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-c-cc.yaml deleted file mode 100644 index 1c8dfd6..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-c-cc.yaml +++ /dev/null @@ -1,40 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dpcon-n2-hmem-48-c -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highmem-48 - maxPodsPerNode: 28 - spot: true - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: n4-highmem-48 - maxPodsPerNode: 28 - spot: true - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n2-highmem-48 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: n4-highmem-48 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-c-co-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-c-co-cc.yaml deleted file mode 100644 index c3cac7c..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-c-co-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dpcon-n2-hmem-48-c-co -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highmem-48 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: n4-highmem-48 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-c-w-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-c-w-cc.yaml deleted file mode 100644 index c7657b8..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2-hmem-48-c-w-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dpcon-n2-hmem-48-c-w -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highmem-32 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: n4-highmem-32 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-16-b-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-16-b-cc.yaml deleted file mode 100644 index fb70799..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-16-b-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dpcon-n2d-hmem-16-b -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highmem-16 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-32-b-od-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-32-b-od-cc.yaml deleted file mode 100644 index aadd027..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-32-b-od-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dpcon-n2d-hmem-32-b-od -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highmem-32 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 200 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-48-a-sensitive-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-48-a-sensitive-cc.yaml deleted file mode 100644 index 0e00279..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-48-a-sensitive-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dpcon-n2d-hmem-48-a-sensitive -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highmem-48 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-48-b-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-48-b-cc.yaml deleted file mode 100644 index 7789fdc..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-48-b-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dpcon-n2d-hmem-48-b -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highmem-48 - maxPodsPerNode: 28 - spot: true - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-48-c-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-48-c-cc.yaml deleted file mode 100644 index 41292e1..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/dpcon-n2d-hmem-48-c-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dpcon-n2d-hmem-48-c -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highmem-48 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/kuberay-operator-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/kuberay-operator-cc.yaml deleted file mode 100644 index 81e4d2e..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/kuberay-operator-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: kuberay-operator -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highmem-48 - maxPodsPerNode: 28 - spot: true - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/megaduo-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/megaduo-cc.yaml deleted file mode 100644 index 792f9e8..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/megaduo-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaduo -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-highcpu-22 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/megaduolite-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/megaduolite-cc.yaml deleted file mode 100644 index 86960ba..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/megaduolite-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaduolite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-custom-16-32768 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/megaquad-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/megaquad-cc.yaml deleted file mode 100644 index 8cf3b9e..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/megaquad-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaquad -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-standard-8 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/megatetra-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/megatetra-cc.yaml deleted file mode 100644 index 7540d63..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/megatetra-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megatetra -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-standard-16 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/n2-hmem-64-ondemand-a-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/n2-hmem-64-ondemand-a-cc.yaml deleted file mode 100644 index dd9d3e8..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/n2-hmem-64-ondemand-a-cc.yaml +++ /dev/null @@ -1,33 +0,0 @@ -# Missing on gke-dataengg-prd-ase1a. Referenced by coordinator of: os-trino-model-bk1 -# Source: NAP pool nap-n2-highmem-64-* (computeclass) on k8s-dataengg-prd-ase1. -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: n2-hmem-64-ondemand-a-cc -spec: - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - activeMigration: - optimizeRulePriority: true - priorities: - - machineType: n2-highmem-64 - spot: false - maxPodsPerNode: 16 - storage: - bootDiskType: pd-ssd - bootDiskSize: 100 - - machineType: n4-highmem-64 - spot: false - maxPodsPerNode: 16 - storage: - bootDiskType: hyperdisk-balanced - bootDiskSize: 100 - priorityDefaults: - location: - zones: - - asia-southeast1-a - - asia-southeast1-b - - asia-southeast1-c - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/n2d-highmem-8-dp-dpcon-zep-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/n2d-highmem-8-dp-dpcon-zep-cc.yaml deleted file mode 100644 index 277d782..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/n2d-highmem-8-dp-dpcon-zep-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: n2d-highmem-8-dp-dpcon-zep -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highmem-8 - maxPodsPerNode: 24 - spot: false - storage: - bootDiskSize: 50 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/n2d-hmem-16-od-ls-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/n2d-hmem-16-od-ls-cc.yaml deleted file mode 100644 index 6bad611..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/n2d-hmem-16-od-ls-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: n2d-hmem-16-od-ls -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highmem-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/n2d-hmem-8-sp-ls-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/n2d-hmem-8-sp-ls-cc.yaml deleted file mode 100644 index 63fa5bc..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/n2d-hmem-8-sp-ls-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: n2d-hmem-8-sp-ls -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highmem-8 - maxPodsPerNode: 18 - spot: true - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/n4-hmem-48-spot-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/n4-hmem-48-spot-cc.yaml deleted file mode 100644 index 4dbb798..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/n4-hmem-48-spot-cc.yaml +++ /dev/null @@ -1,27 +0,0 @@ -# Missing on gke-dataengg-prd-ase1a. Referenced by workers of: -# os-prismsdk-abservice, os-trino-mb-s2, oss-trino-mb-pri-bk1, oss-trino-mb-pri-bk2 -# Source: NAP pool nap-n4-highmem-48-spot-* (computeclass) on k8s-dataengg-prd-ase1. -# N4 machines require hyperdisk-balanced boot disk. -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: n4-hmem-48-spot-cc -spec: - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - activeMigration: - optimizeRulePriority: true - priorities: - - machineType: n4-highmem-48 - spot: true - maxPodsPerNode: 16 - storage: - bootDiskType: hyperdisk-balanced - bootDiskSize: 100 - priorityDefaults: - location: - zones: - - asia-southeast1-a - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/nginx-shared-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/nginx-shared-cc.yaml deleted file mode 100644 index bb15dfe..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/nginx-shared-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: nginx-shared -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-highcpu-4 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - - machineType: c4-highcpu-4 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/np-dp-kuberay-4c-16g-prd-ase1-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/np-dp-kuberay-4c-16g-prd-ase1-cc.yaml deleted file mode 100644 index 0101d83..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/np-dp-kuberay-4c-16g-prd-ase1-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: np-dp-kuberay-4c-16g-prd-ase1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-standard-4 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/os-trino-poc-n2d-dp-dpcon-wk-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/os-trino-poc-n2d-dp-dpcon-wk-cc.yaml deleted file mode 100644 index 46c84ea..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/os-trino-poc-n2d-dp-dpcon-wk-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: os-trino-poc-n2d-dp-dpcon-wk -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highmem-48 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 200 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/strimzi-kafka-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/strimzi-kafka-cc.yaml deleted file mode 100644 index dd5903a..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/strimzi-kafka-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: strimzi-kafka -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highmem-4 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 50 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/sumoduo-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/sumoduo-cc.yaml deleted file mode 100644 index 446b8ec..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/sumoduo-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumoduo -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-highcpu-44 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: c4-highcpu-48 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/sumoduolite-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/sumoduolite-cc.yaml deleted file mode 100644 index 49a97a8..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/sumoduolite-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumoduolite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-custom-32-65536 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/sumounolite-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/sumounolite-cc.yaml deleted file mode 100644 index 29bdd1c..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/sumounolite-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumounolite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highcpu-32 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: n4-highcpu-32 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/vmagent-mds-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/vmagent-mds-cc.yaml deleted file mode 100644 index f01b431..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/vmagent-mds-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmagent-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/vminsert-mds-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/vminsert-mds-cc.yaml deleted file mode 100644 index eb69b9f..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/vminsert-mds-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vminsert-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-highcpu-16 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/vmselect-mds-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/vmselect-mds-cc.yaml deleted file mode 100644 index 9c9d977..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/vmselect-mds-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmselect-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-highcpu-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/vmstack-startree-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/vmstack-startree-cc.yaml deleted file mode 100644 index fdfa17c..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/vmstack-startree-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstack-startree -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml deleted file mode 100644 index 09c35f1..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage-n4d -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highmem-32 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/warpstream-dp-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/warpstream-dp-cc.yaml deleted file mode 100644 index dadb8d1..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/warpstream-dp-cc.yaml +++ /dev/null @@ -1,23 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: warpstream-dp -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-standard-8 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - # N4 machine family requires Hyperdisk (pd-ssd is not supported on N4). - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/wk-trino-spot-48c-384g-prd-ase1-cc.yaml b/helm-overrides/gke-dataengg-prd-ase1a/computeclass/wk-trino-spot-48c-384g-prd-ase1-cc.yaml deleted file mode 100644 index 2688165..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/computeclass/wk-trino-spot-48c-384g-prd-ase1-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: wk-trino-spot-48c-384g-prd-ase1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-deng-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highmem-48 - maxPodsPerNode: 28 - spot: true - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: n4-highmem-48 - maxPodsPerNode: 28 - spot: true - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dataengg-prd-ase1a/conntrack-adjuster/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 02c4a82..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,17 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - max: 2097152 - hashsize: 524288 - sleepInterval: 30 - -nodeAffinity: - enabled: true - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: In - values: - - "contour-external" - - "contour-internal-0" diff --git a/helm-overrides/gke-dataengg-prd-ase1a/contour-ca-issuer/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/contour-ca-issuer/custom-values.yaml deleted file mode 100644 index af632b8..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/contour-ca-issuer/custom-values.yaml +++ /dev/null @@ -1,17 +0,0 @@ -# ClusterIssuer override for gke-dataengg-prd-ase1a (prd dataengg cluster, ase1a zone). -# ClusterIssuer is a cluster-scoped resource — this file pins the issuer -# identity for this cluster so the name is auditable per-cluster. -# -# Must match the `issuerRef.name` in consuming Certificate CRs (see -# devops-helm-charts/2.0.0/templates/proxyless-grpc-cert.yaml). - -issuerName: contour-dataengg-prd-ca-issuer -rootCASecretName: contour-dataengg-ca - -externalSecret: - enabled: true - vaultPath: meesho/devops/contour/root-ca - refreshInterval: "0" - secretStoreRef: - name: vault-backend -namespace: cert-manager-dataengg-prd \ No newline at end of file diff --git a/helm-overrides/gke-dataengg-prd-ase1a/contour-cert-checker/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/contour-cert-checker/custom-values.yaml deleted file mode 100644 index 4bba335..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=gke-dataengg-prd-ase1a"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - cloud.google.com/compute-class: dataengg-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: dataengg-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/gke-dataengg-prd-ase1a/contour-external/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/contour-external/custom-values.yaml deleted file mode 100644 index 86aacd7..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/contour-external/custom-values.yaml +++ /dev/null @@ -1,114 +0,0 @@ -fullnameOverride: "contour-external-dataengg-prd" -configInline: - enableExternalNameService: true - network: - num-trusted-hops: 1 - timeouts: - connection-idle-timeout: 305s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 2Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-shared-cc-v1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-external-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-external-cc-v1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "40" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 6' - resources: - requests: - cpu: 6 - memory: 8Gi - limits: - cpu: 6 - memory: 12Gi - service: - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-dataengg-a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-dataengg-prd-ase1a/contour-internal-0/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/contour-internal-0/custom-values.yaml deleted file mode 100644 index 4998f54..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,117 +0,0 @@ -fullnameOverride: "contour-internal-0-dataengg-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 2Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-shared-cc-v1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-0-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-internal-0-cc-v1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 6' - resources: - requests: - cpu: 6 - memory: 3Gi - limits: - cpu: 6 - memory: 12Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-dataengg-a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-dataengg-prd-ase1a/contour-internal-1/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/contour-internal-1/custom-values.yaml deleted file mode 100644 index b5e0aae..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,117 +0,0 @@ -fullnameOverride: "contour-internal-1-dataengg-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 2Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-shared-cc-v1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-1-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-internal-1-cc-v1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 30' - resources: - requests: - cpu: 30 - memory: 16Gi - limits: - cpu: 30 - memory: 57Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int1-dataengg-a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-dataengg-prd-ase1a/contour-internal-intra-0/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index d73f5c7..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,114 +0,0 @@ -fullnameOverride: "contour-internal-0-dataengg-intra-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 2Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-shared-cc-v1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-intra-0-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-intra-0-cc-v1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 14' - resources: - requests: - cpu: 14 - memory: 3Gi - limits: - cpu: 14 - memory: 27Gi - service: - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-dataengg-prd-ase1a/contour-internal-intra-1/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index 1c68927..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,114 +0,0 @@ -fullnameOverride: "contour-internal-1-dataengg-intra-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 2Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-shared-cc-v1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-intra-1-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-intra-1-cc-v1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 14' - resources: - requests: - cpu: 14 - memory: 8Gi - limits: - cpu: 14 - memory: 27Gi - service: - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-dataengg-prd-ase1a/coredns/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/coredns/custom-values.yaml deleted file mode 100644 index 4a114e5..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/coredns/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -replicaCount: 30 -communicationType: "intra" - -labels: - bu: dataengg - team: dataengg-devops - env: prd - -clusterIP: "10.1.32.2" - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "dataengg-devops" - effect: "NoSchedule" - -nodeSelector: - cloud.google.com/compute-class: dataengg-devops - kubernetes.io/os: linux \ No newline at end of file diff --git a/helm-overrides/gke-dataengg-prd-ase1a/external-secrets/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/external-secrets/custom-values.yaml deleted file mode 100644 index e5bd948..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/external-secrets/custom-values.yaml +++ /dev/null @@ -1,61 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Annotations to add to Pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: dataengg-devops - nodeSelector: - cloud.google.com/compute-class: dataengg-devops - certController: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: dataengg-devops - nodeSelector: - cloud.google.com/compute-class: dataengg-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - webhook: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: dataengg-devops - nodeSelector: - cloud.google.com/compute-class: dataengg-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" \ No newline at end of file diff --git a/helm-overrides/gke-dataengg-prd-ase1a/flagger/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/flagger/custom-values.yaml deleted file mode 100644 index d9c3ac0..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/flagger/custom-values.yaml +++ /dev/null @@ -1,62 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "512Mi" - cpu: "1000m" - requests: - memory: "256Mi" - cpu: "100m" - -nodeSelector: - cloud.google.com/compute-class: dataengg-devops - -tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: dataengg-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: dataengg-devops - bu: infra - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-dataengg-rollout-service.prd-dataengg-rollout-service.svc.cluster.local" - -rolloutService: - enabled: true - apps: \ No newline at end of file diff --git a/helm-overrides/gke-dataengg-prd-ase1a/fluentd/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/fluentd/custom-values.yaml deleted file mode 100644 index 0799eb8..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/fluentd/custom-values.yaml +++ /dev/null @@ -1,730 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-deng-desre-fluentd-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: dataengg - team: sre - type: fluentd - service: fluentd-dataengg-a-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: dataengg - team: sre - type: fluentd - service: fluentd-dataengg-a-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -# - secretRef: -# name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: - - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - from_encoding "utf-8" - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/gke-dataengg-prd-ase1a/ingress-nginx-external/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/ingress-nginx-external/custom-values.yaml deleted file mode 100644 index 65fa0e0..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/ingress-nginx-external/custom-values.yaml +++ /dev/null @@ -1,34 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - resources: - requests: - cpu: 100m - memory: 768Mi - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 120 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 70 - ingressClass: nginx-external - ingressClassByName: true - watchIngressWithoutClass: false - ingressClassResource: - controllerValue: k8s.io/ingress-nginx-external - name: nginx-external - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dataengg-external-a-prd"}}}' - nodeSelector: - cloud.google.com/compute-class: nginx-shared - tolerations: - - key: cloud.google.com/compute-class - operator: Equal - value: nginx-shared - effect: NoSchedule diff --git a/helm-overrides/gke-dataengg-prd-ase1a/ingress-nginx-internal/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/ingress-nginx-internal/custom-values.yaml deleted file mode 100644 index de3475b..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/ingress-nginx-internal/custom-values.yaml +++ /dev/null @@ -1,33 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - resources: - requests: - cpu: 100m - memory: 768Mi - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 120 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 70 - extraArgs: - enable-ssl-passthrough: true - tcp-services-configmap: nginx-dataengg-internal/tcp-services - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dataengg-internal-a-prd"}}}' - nodeSelector: - cloud.google.com/compute-class: nginx-shared - tolerations: - - key: cloud.google.com/compute-class - operator: Equal - value: nginx-shared - effect: NoSchedule diff --git a/helm-overrides/gke-dataengg-prd-ase1a/ingress-nginx-secured/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/ingress-nginx-secured/custom-values.yaml deleted file mode 100644 index 1f65d93..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/ingress-nginx-secured/custom-values.yaml +++ /dev/null @@ -1,33 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - resources: - requests: - cpu: 100m - memory: 768Mi - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 120 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 70 - extraArgs: - enable-ssl-passthrough: true - tcp-services-configmap: nginx-dataengg-secured/tcp-services - ingressClassResource: - name: nginx-secured - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dataengg-secured-a-prd"}}}' - nodeSelector: - cloud.google.com/compute-class: nginx-shared - tolerations: - - key: cloud.google.com/compute-class - operator: Equal - value: nginx-shared - effect: NoSchedule diff --git a/helm-overrides/gke-dataengg-prd-ase1a/keda/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/keda/custom-values.yaml deleted file mode 100644 index 5a75434..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/keda/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - cloud.google.com/compute-class: dataengg-devops - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "dataengg-devops" - effect: "NoSchedule" - podLabels: - bu: "dataengg" - team: "dataengg-devops" - metricsAdapter: - bu: "dataengg" - team: "dataengg-devops" - resources: - webhooks: - limits: - cpu: 50m - memory: 100Mi - requests: - cpu: 10m - memory: 20Mi \ No newline at end of file diff --git a/helm-overrides/gke-dataengg-prd-ase1a/kube-dns/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/kube-dns/custom-values.yaml deleted file mode 100644 index 018f0eb..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.1.32.2"],"prd.mrouter.int.svc.cluster.local":["10.1.32.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.1.32.2"]} diff --git a/helm-overrides/gke-dataengg-prd-ase1a/kube-events/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/kube-events/custom-values.yaml deleted file mode 100644 index 4b4335b..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-dataengg-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - cloud.google.com/compute-class: dataengg-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: dataengg-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: dataengg-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: dataengg-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: dataengg-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: dataengg-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/gke-dataengg-prd-ase1a/kube-state-metrics/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 1211ec5..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,427 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: "" - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-dataengg-a-prd - -imagePullSecrets: [] -# - name: "image-pull-secret" - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "kube-state-metrics-dataengg-a-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - cloud.google.com/compute-class: dataengg-devops -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: dataengg-devops - operator: Equal -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 200m - memory: 1000Mi - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/gke-dataengg-prd-ase1a/kubectl-mcp-server/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/kubectl-mcp-server/custom-values.yaml deleted file mode 100644 index d0edd21..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/kubectl-mcp-server/custom-values.yaml +++ /dev/null @@ -1,119 +0,0 @@ -fullnameOverride: "kubectl-mcp-server" - -replicas: 1 - -image: - # In-house hardened build (helm-templates/kubectl-mcp-server/docker/) — - # NOT the upstream Docker Hub image. 266→18 HIGH/CRIT vulns, non-root, - # multi-stage slim base, kubectl v1.33.12 / helm v3.21.0. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/prd/devop/kubectl-mcp-server - tag: "v2" - pullPolicy: IfNotPresent - -labels: - bu: infra - team: devops - service: kubectl-mcp-server - env: prd - -serviceAccount: - create: true - annotations: {} - -rbac: - create: true - -podAnnotations: {} - -podSecurityContext: {} - -securityContext: - runAsNonRoot: true - runAsUser: 1000 - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -priorityClassName: "" - -nodeSelector: - cloud.google.com/compute-class: "dataengg-devops" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "dataengg-devops" - effect: NoSchedule - -affinity: {} - -resources: - requests: - cpu: 250m - memory: 512Mi - limits: - cpu: 1000m - memory: 1Gi - -mcp: - mode: single - transport: http - host: "0.0.0.0" - port: 8000 - -auth: - allowAnonymous: false - -externalSecrets: - enabled: true - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/prd/cntr/devop/kubectl-mcp-server-dataengg" - -# tcpSocket on purpose — streamable-http transport exposes only /mcp, -# there is no /health route (see chart values.yaml note). -livenessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - -readinessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - -service: - port: 8000 - type: ClusterIP - -# Contour standard (2.0.0-style): renders HTTPProxy parent/child + -intra -# instead of a networking.k8s.io Ingress. -createContourGateway: true - -ingress: - enabled: true - # dataengg prd exposes the -0 contour internal classes (no -1); the chart - # auto-derives contour-internal-intra-0. Verified on gke-dataengg-prd-ase1a. - ingressClassName: contour-internal-0 - servicePortNumber: 8000 - hosts: - - host: kubectl-mcp-server-dataengg.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 diff --git a/helm-overrides/gke-dataengg-prd-ase1a/kyverno/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/kyverno/custom-values.yaml deleted file mode 100644 index 96d3e3c..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/kyverno/custom-values.yaml +++ /dev/null @@ -1,2243 +0,0 @@ -global: - - # -- Internal settings used with `helm template` to generate install manifest - # @ignored - templating: - enabled: false - debug: false - version: ~ - - image: - # -- (string) Global value that allows to set a single image registry across all deployments. - # When set, it will override any values set under `.image.registry` across the chart. - registry: ~ - # -- (list) Global list of Image pull secrets - # When set, it will override any values set under `imagePullSecrets` under different components across the chart. - imagePullSecrets: [] - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - caCertificates: - # -- Global CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - # Individual controller values will override this global value - data: ~ - - # -- Global value to set single volume to be mounted for CA certificates for all deployments. - # Not used when `.Values.global.caCertificates.data` is defined - # Individual controller values will override this global value - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Additional container environment variables to apply to all containers and init containers - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Global node labels for pod assignment. Non-global values will override the global value. - nodeSelector: - dedicated: dataengg-kyverno - - # -- Global List of node taints to tolerate. Non-global values will override the global value. - tolerations: - - key: "dedicated" - operator: "Equal" - value: "dataengg-kyverno" - effect: "NoSchedule" - -# -- (string) Override the name of the chart -nameOverride: ~ - -# -- (string) Override the expanded name of the chart -fullnameOverride: ~ - -# -- (string) Override the namespace the chart deploys to -namespaceOverride: ~ - -upgrade: - # -- Upgrading from v2 to v3 is not allowed by default, set this to true once changes have been reviewed. - fromV2: false - -apiVersionOverride: - # -- (string) Override api version used to create `PodDisruptionBudget`` resources. - # When not specified the chart will check if `policy/v1/PodDisruptionBudget` is available to - # determine the api version automatically. - podDisruptionBudget: ~ - -rbac: - roles: - # -- Aggregate ClusterRoles to Kubernetes default user-facing roles. For more information, see [User-facing roles](https://kubernetes.io/docs/reference/access-authn-authz/rbac/#user-facing-roles) - aggregate: - admin: true - view: true - -# Use openreports.io as the API group for reporting -openreports: - # -- Enable OpenReports feature in controllers - enabled: false - # -- Whether to install CRDs from the upstream OpenReports chart. Setting this to true requires enabled to also be true. - installCrds: false - -# CRDs configuration -crds: - - # -- Whether to have Helm install the Kyverno CRDs, if the CRDs are not installed by Helm, they must be added before policies can be created - install: true - - reportsServer: - # -- Kyverno reports-server is used in your cluster - enabled: false - - groups: - - # -- Install CRDs in group `kyverno.io` - kyverno: - cleanuppolicies: true - clustercleanuppolicies: true - clusterpolicies: true - globalcontextentries: true - policies: true - policyexceptions: true - updaterequests: true - - # -- Install CRDs in group `policies.kyverno.io` - policies: - validatingpolicies: true - policyexceptions: true - imagevalidatingpolicies: true - namespacedimagevalidatingpolicies: true - mutatingpolicies: true - generatingpolicies: true - deletingpolicies: true - namespaceddeletingpolicies: true - namespacedvalidatingpolicies: true - - # -- Install CRDs in group `reports.kyverno.io` - reports: - clusterephemeralreports: true - ephemeralreports: true - - # -- Install CRDs in group `wgpolicyk8s.io` - wgpolicyk8s: - clusterpolicyreports: false ##lk - policyreports: false ##lk - - # -- Additional CRDs annotations (Replace=true avoids 262144-byte annotation limit on large CRDs) - annotations: - argocd.argoproj.io/sync-options: Replace=true - # strategy.spinnaker.io/replace: 'true' - - # -- Additional CRDs labels - customLabels: {} - - migration: - - # -- Enable CRDs migration using helm post upgrade hook - enabled: true - - # -- Resources to migrate - resources: - - cleanuppolicies.kyverno.io - - clustercleanuppolicies.kyverno.io - - clusterpolicies.kyverno.io - - globalcontextentries.kyverno.io - - policies.kyverno.io - - policyexceptions.kyverno.io - - updaterequests.kyverno.io - - deletingpolicies.policies.kyverno.io - - generatingpolicies.policies.kyverno.io - - imagevalidatingpolicies.policies.kyverno.io - - namespacedimagevalidatingpolicies.policies.kyverno.io - - mutatingpolicies.policies.kyverno.io - - namespaceddeletingpolicies.policies.kyverno.io - - namespacedvalidatingpolicies.policies.kyverno.io - - policyexceptions.policies.kyverno.io - - validatingpolicies.policies.kyverno.io - - image: - # -- (string) Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- (string) Image repository - repository: kyverno/kyverno-cli - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- (string) Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podResources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -# Configuration -config: - - # -- Create the configmap. - create: true - - # -- Preserve the configmap settings during upgrade. - preserve: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - # -- Enable registry mutation for container images. Enabled by default. - enableDefaultRegistryMutation: true - - # -- The registry hostname used for the image mutation. - defaultRegistry: docker.io - - # -- Exclude groups - excludeGroups: - - system:nodes - - # -- Exclude usernames - excludeUsernames: [] - # - '!system:kube-scheduler' - - # -- Exclude roles - excludeRoles: [] - - # -- Exclude roles - excludeClusterRoles: [] - - # -- Generate success events. - generateSuccessEvents: false - - # -- Resource types to be skipped by the Kyverno policy engine. - # Make sure to surround each entry in quotes so that it doesn't get parsed as a nested YAML list. - # These are joined together without spaces, run through `tpl`, and the result is set in the config map. - # @default -- See [values.yaml](values.yaml) - resourceFilters: - - '[Event,*,*]' - # - '[*/*,kube-system,*]' - - '[*/*,kube-public,*]' - - '[*/*,kube-node-lease,*]' - - '[Node,*,*]' - - '[Node/?*,*,*]' - - '[APIService,*,*]' - - '[APIService/?*,*,*]' - - '[TokenReview,*,*]' - - '[SubjectAccessReview,*,*]' - - '[SelfSubjectAccessReview,*,*]' - - '[Binding,*,*]' - - '[Pod/binding,*,*]' - - '[ReplicaSet,*,*]' - - '[ReplicaSet/?*,*,*]' - - '[EphemeralReport,*,*]' - - '[ClusterEphemeralReport,*,*]' - # exclude resources from the chart - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:additional]' - - '[ClusterRoleBinding,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.configMapName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.metricsConfigMapName" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Job,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[Job/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.admission-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.background-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - # -- Sets the threshold for the total number of UpdateRequests generated for mutateExisitng and generate policies. - updateRequestThreshold: 1000 - - # -- Defines the `namespaceSelector`/`objectSelector` in the webhook configurations. - # The Kyverno namespace is excluded if `excludeKyvernoNamespace` is `true` (default) - webhooks: - # Exclude namespaces - # namespaceSelector: lk-s2 - # matchExpressions: - # - key: kubernetes.io/metadata.name - # operator: NotIn - # values: - # - kube-systemm lk-s2 - # Exclude objects - # objectSelector: - # matchExpressions: - # - key: webhooks.kyverno.io/exclude - # operator: DoesNotExist - - # -- Defines annotations to set on webhook configurations. - webhookAnnotations: - # Example to disable admission enforcer on AKS: - # 'admissions.enforcer/disabled': 'true' lk-s3 - - # -- Defines labels to set on webhook configurations. - webhookLabels: {} - # Example to adopt webhook resources in ArgoCD: - # 'argocd.argoproj.io/instance': 'kyverno' - - # -- Defines match conditions to set on webhook configurations (requires Kubernetes 1.27+). - matchConditions: [] - - # -- Exclude Kyverno namespace - # Determines if default Kyverno namespace exclusion is enabled for webhooks and resourceFilters - excludeKyvernoNamespace: true - - # -- resourceFilter namespace exclude - # Namespaces to exclude from the default resourceFilters - resourceFiltersExcludeNamespaces: [] - - # -- resourceFilters exclude list - # Items to exclude from config.resourceFilters - resourceFiltersExclude: [] - - # -- resourceFilter namespace include - # Namespaces to include to the default resourceFilters - resourceFiltersIncludeNamespaces: [] - - # -- resourceFilters include list - # Items to include to config.resourceFilters - resourceFiltersInclude: [] - -# Metrics configuration -metricsConfig: - - # -- Create the configmap. - create: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - namespaces: - - # -- List of namespaces to capture metrics for. - include: [] - - # -- list of namespaces to NOT capture metrics for. - exclude: [] - - # -- (string) Rate at which metrics should reset so as to clean up the memory footprint of kyverno metrics, if you might be expecting high memory footprint of Kyverno's metrics. Default: 0, no refresh of metrics. WARNING: This flag is not working since Kyverno 1.8.0 - metricsRefreshInterval: ~ - # metricsRefreshInterval: 24h - - # -- (list) Configures the bucket boundaries for all Histogram metrics, changing this configuration requires restart of the kyverno admission controller - bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 15, 20, 25, 30] - - # -- (map) Configures the exposure of individual metrics, by default all metrics and all labels are exported, changing this configuration requires restart of the kyverno admission controller - metricsExposure: - kyverno_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_image_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_mutating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_generating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_admission_review_duration_seconds: - # enabled: false - disabledLabelDimensions: ["resource_namespace"] - kyverno_policy_rule_info_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_policy_results_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_admission_requests_total: - disabledLabelDimensions: ["resource_namespace"] - kyverno_cleanup_controller_deletedobjects_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - -# -- Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -imagePullSecrets: {} - # regcred: - # registry: foo.example.com - # username: foobar - # password: secret - # regcred2: - # registry: bar.example.com - # username: barbaz - # password: secret2 - -# -- Existing Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -existingImagePullSecrets: [] - # - test-registry - # - other-test-registry - -# Tests configuration -test: - # -- Sleep time before running test - sleep: 20 - - image: - # -- (string) Image registry - registry: curlimages - # -- Image repository - repository: curl - # -- Image tag - # Defaults to `latest` if omitted - tag: '8.10.1' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - # -- Security context for the test containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- Additional Pod annotations - podAnnotations: {} - - # -- List of node taints to tolerate - tolerations: [] - -# -- Additional labels -customLabels: {} - - -webhooksCleanup: - # -- Create a helm pre-delete hook to cleanup webhooks. - enabled: true - - autoDeleteWebhooks: - # -- Allow webhooks controller to delete webhooks using finalizers - enabled: false - - image: - # -- (string) Image registry - registry: registry.k8s.io - # -- Image repository - repository: kubectl - # -- Image tag - # Defaults to `latest` if omitted - tag: 'v1.32.7' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -grafana: - # -- Enable grafana dashboard creation. - enabled: false - - # -- Configmap name template. - configMapName: '{{ include "kyverno.fullname" . }}-grafana' - - # -- (string) Namespace to create the grafana dashboard configmap. - # If not set, it will be created in the same namespace where the chart is deployed. - namespace: ~ - - # -- Grafana dashboard configmap annotations. - annotations: {} - - # -- Grafana dashboard configmap labels - labels: - grafana_dashboard: "1" - - # -- create GrafanaDashboard custom resource referencing to the configMap. - # according to https://grafana-operator.github.io/grafana-operator/docs/examples/dashboard_from_configmap/readme/ - grafanaDashboard: - create: false - folder: kyverno - allowCrossNamespaceImport: true - matchLabels: - dashboards: "grafana" - -# Features configuration -features: - admissionReports: - # -- Enables the feature - enabled: true - aggregateReports: - # -- Enables the feature - enabled: true - policyReports: - # -- Enables the feature - enabled: false ##lk - validatingAdmissionPolicyReports: - # -- Enables the feature - enabled: true - mutatingAdmissionPolicyReports: - # -- Enables the feature - enabled: false - reporting: - # -- Enables the feature - validate: true - # -- Enables the feature - mutate: true - # -- Enables the feature - mutateExisting: true - # -- Enables the feature - imageVerify: true - # -- Enables the feature - generate: true - autoUpdateWebhooks: - # -- Enables the feature - enabled: true - backgroundScan: - # -- Enables the feature - enabled: true - # -- Number of background scan workers - backgroundScanWorkers: 2 - # -- Background scan interval - backgroundScanInterval: 1h - # -- Skips resource filters in background scan - skipResourceFilters: true - configMapCaching: - # -- Enables the feature - enabled: true - controllerRuntimeMetrics: - # -- Bind address for controller-runtime metrics (use "0" to disable it) - bindAddress: ":8080" - deferredLoading: - # -- Enables the feature - enabled: true - dumpPayload: - # -- Enables the feature - enabled: false - forceFailurePolicyIgnore: - # -- Enables the feature - enabled: false - generateValidatingAdmissionPolicy: - # -- Enables the feature - enabled: true - generateMutatingAdmissionPolicy: - # -- Enables the feature - enabled: false - dumpPatches: - # -- Enables the feature - enabled: false - globalContext: - # -- Maximum allowed response size from API Calls. A value of 0 bypasses checks (not recommended) - maxApiCallResponseLength: 2000000 - logging: - # -- Logging format - format: text - # -- Logging verbosity - verbosity: 2 - omitEvents: - # -- Events which should not be emitted (possible values `PolicyViolation`, `PolicyApplied`, `PolicyError`, and `PolicySkipped`) - eventTypes: - - PolicyApplied - - PolicySkipped - # - PolicyViolation - # - PolicyError - policyExceptions: - # -- Enables the feature - enabled: true - # -- Restrict policy exceptions to a single namespace - # Set to "*" to allow exceptions in all namespaces - namespace: '' - protectManagedResources: - # -- Enables the feature - enabled: false - registryClient: - # -- Allow insecure registry - allowInsecure: false - # -- Enable registry client helpers - credentialHelpers: - - default - - google - - amazon - - azure - - github - ttlController: - # -- Reconciliation interval for the label based cleanup manager - reconciliationInterval: 1m - tuf: - # -- Enables the feature - enabled: false - # -- (string) Path to Tuf root - root: ~ - # -- (string) Raw Tuf root - rootRaw: ~ - # -- (string) Tuf mirror - mirror: ~ - -# Admission controller configuration -admissionController: - autoscaling: - # -- Enable horizontal pod autoscaling - enabled: true - - # -- Minimum number of pods - minReplicas: 1 - - # -- Maximum number of pods - maxReplicas: 10 - - # -- Target CPU utilization percentage - targetCPUUtilizationPercentage: 80 - - # -- Configurable scaling behavior - behavior: {} - - # -- Overrides features defined at the root level - featuresOverride: - admissionReports: - # -- Max number of admission reports allowed in flight until the admission controller stops creating new ones - backPressureThreshold: 1000 - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- The ServiceAccount name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno admission controller activities. - # This will help ensure Kyverno stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- admissionController webhook server port - # in case you are using hostNetwork: true, you might want to change the port the webhookServer is listening to - webhookServer: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - admission-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.admissionController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - initContainer: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyvernopre - # -- (string) Image tag - # If missing, defaults to image.tag - tag: ~ - # -- (string) Image pull policy - # If missing, defaults to image.pullPolicy - pullPolicy: ~ - - resources: - # -- Pod resource limits - limits: - cpu: 1000m - memory: 1Gi - # -- Pod resource requests - requests: - cpu: 1000m - memory: 1Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - container: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyverno - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - resources: - # -- Pod resource limits - limits: - cpu: 1900m - memory: 3Gi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Array of extra init containers - extraInitContainers: [] - # - name: init-container - # image: busybox - # command: ['sh', '-c', 'echo Hello'] - - # -- Array of extra containers to run alongside kyverno - extraContainers: [] - # - name: myapp-container - # image: busybox - # command: ['sh', '-c', 'echo Hello && sleep 3600'] - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Kyverno's metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Background controller configuration -backgroundController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable background controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: - - apiGroups: - - networking.k8s.io - resources: - - ingresses - - ingressclasses - - networkpolicies - verbs: - - create - - update - - patch - - delete - - apiGroups: - - rbac.authorization.k8s.io - resources: - - rolebindings - - roles - verbs: - - create - - update - - patch - - delete - - apiGroups: - - '' - resources: - - configmaps - - resourcequotas - - limitranges - verbs: - - create - - update - - patch - - delete - - apiGroups: - - resource.k8s.io - resources: - - resourceclaims - - resourceclaimtemplates - verbs: - - create - - delete - - update - - patch - - deletecollection - - apiGroups: - - autoscaling - resources: - - horizontalpodautoscalers - verbs: - - get - - list - - watch - - update - - patch - - apiGroups: - - apps - resources: - - deployments - - statefulsets - - replicasets - verbs: - - get - - list - - watch - - update - - patch - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - # - patch - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/background-controller - # -- Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 1900m - memory: 2Gi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - background-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.backgroundController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - # -- backgroundController server port - # in case you are using hostNetwork: true, you might want to change the port the backgroundController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Cleanup controller configuration -cleanupController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable cleanup controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - delete - # - list - # - watch - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/cleanup-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- cleanupController server port - # in case you are using hostNetwork: true, you might want to change the port the cleanupController is listening to - server: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 1100m - memory: 1.2Gi - # -- Pod resource requests - requests: - cpu: 1100m - memory: 1.2Gi - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - cleanup-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `service.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Reports controller configuration -reportsController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable reports controller. - enabled: false ##lk - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/reports-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno reports controller activities. - # This will help ensure Kyverno reports stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - memory: 128Mi - # -- Pod resource requests - requests: - cpu: 100m - memory: 64Mi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - reports-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.reportsController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- (string) Service node port. - # Only used if `type` is `NodePort`. - nodePort: ~ - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- (string) Traces receiver address - address: ~ - # -- (string) Traces receiver port - port: ~ - # -- (string) Traces receiver credentials - creds: ~ - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- (string) Otel collector endpoint - collector: ~ - # -- (string) Otel collector credentials - creds: ~ - - # -- reportsController server port - # in case you are using hostNetwork: true, you might want to change the port the reportsController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - - # -- Enable sanity check for reports CRDs - sanityChecks: true diff --git a/helm-overrides/gke-dataengg-prd-ase1a/kyverno/policies/protect-namespace.yaml b/helm-overrides/gke-dataengg-prd-ase1a/kyverno/policies/protect-namespace.yaml deleted file mode 100644 index 7849d2c..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/kyverno/policies/protect-namespace.yaml +++ /dev/null @@ -1,40 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: protect-critical-namespaces -spec: - background: false - rules: - - name: deny-deletion-of-critical-namespaces - match: - any: - - resources: - kinds: - - Namespace - names: - - keda-dataengg-prd - - contour-internal-0-dataengg-prd - - contour-internal-0-dataengg-prd-intra - - contour-external-dataengg-prd - - external-secrets-dataengg-prd - - flagger-dataengg-prd - - victoriametrics - - kube-system - - monitoring - - telegraf-operator - - loadtester - - opentelemetry - - fluentd - - kube-events - - observability - - contour-cert-checker-ns - - aurva-dataplane - validate: - failureAction: Enforce - message: "Deletion of critical namespaces is not allowed." - deny: - conditions: - any: - - key: "{{request.operation}}" - operator: Equals - value: DELETE diff --git a/helm-overrides/gke-dataengg-prd-ase1a/kyverno/policies/restrict-replicas.yaml b/helm-overrides/gke-dataengg-prd-ase1a/kyverno/policies/restrict-replicas.yaml deleted file mode 100644 index 101bbeb..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/kyverno/policies/restrict-replicas.yaml +++ /dev/null @@ -1,29 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: restrict-minimum-replicas-selected-ns -spec: - background: false - rules: - - name: min-replicas-deployments-statefulsets - match: - resources: - kinds: - - Deployment - - StatefulSet - - Deployment/scale - - StatefulSet/scale - names: - - coredns - namespaces: - - kube-system - operations: - - CREATE - - UPDATE - validate: - failureAction: Enforce - message: "Deployments and StatefulSets must have at least 5 replicas in this namespace." - anyPattern: - # Case 1: replicas explicitly set and >= 5 - - spec: - replicas: ">=5" diff --git a/helm-overrides/gke-dataengg-prd-ase1a/loadtester/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/loadtester/custom-values.yaml deleted file mode 100644 index 54d6e6e..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/loadtester/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -replicaCount: 1 - -Namespace: loadtester - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/dev/devops/flagger-loadtester - tag: 0.30.0 - pullPolicy: IfNotPresent - pullSecret: - -podLabels: - bu: dataengg - team: dataengg-shared - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - -podPriorityClassName: "" - -logLevel: info -cmd: - timeout: 1h - namespaceRegexp: "" - -nameOverride: "loadtester" -fullnameOverride: "" - -env: [] - -service: - type: ClusterIP - port: 80 - -resources: - requests: - cpu: 10m - memory: 64Mi - -volumes: [] -volumeMounts: [] - -nodeSelector: - cloud.google.com/compute-class: dataengg-devops - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "dataengg-devops" - effect: "NoSchedule" - -affinity: {} - -ingress: - slowStart: - enabled: false - enabled: true - ingressClassName: contour-internal-0 - servicePort: http - hosts: - - host: dataengg-flagger-loadtester.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / -createContourGateway: true -contourResponseTimeout: false -rbac: - # rbac.create: `true` if rbac resources should be created - create: true - # rbac.scope: `cluster` to create cluster-scope rbac resources (ClusterRole/ClusterRoleBinding) - # otherwise, namespace-scope rbac resources will be created (Role/RoleBinding) - scope: - # rbac.rules: array of rules to apply to the role. example: - # rules: - # - apiGroups: [""] - # resources: ["pods"] - # verbs: ["list", "get"] - rules: [] - -# name of an existing service account to use - if not creating rbac resources -serviceAccountName: "" - -# App Mesh virtual node settings (to be used for AppMesh v1beta1) -meshName: "" -#backends: -# - app1.namespace -# - app2.namespace - -# App Mesh virtual node settings (to be used for AppMesh v1beta2) -appmesh: - enabled: false - backends: - - podinfo - - podinfo-canary - -#Istio virtual service and gatway settings. TLS secrets should be in namespace before enbaled it. ( secret format loadtester.fullname ) -istio: - enabled: false - host: flagger-loadtester.flagger - gateway: - enabled: false - tls: - enabled: false - httpsRedirect: false - -# when enabled, it will add a security context for the loadtester pod -securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 100 - runAsGroup: 101 - -podDisruptionBudget: - enabled: false - minAvailable: 1 diff --git a/helm-overrides/gke-dataengg-prd-ase1a/opentelemetry-coralogix/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/opentelemetry-coralogix/custom-values.yaml deleted file mode 100644 index 249fd4f..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/opentelemetry-coralogix/custom-values.yaml +++ /dev/null @@ -1,281 +0,0 @@ -global: - domain: "coralogixsg.com" - defaultApplicationName: "default" - defaultSubsystemName: "nodes" - - # Old endpoint based configuration, - # please use domain instead. - traces: - endpoint: "" - metrics: - endpoint: "" - logs: - endpoint: "" - -# set distribution to openshift for openshift clusters -distribution: "" - -# Opentelemetry collector configuration -dedicatedValue: false -schedulerName: default-scheduler - -labels: - bu: "dataengg" - team: "dataengg-sre" - service: "opentelemetry-dataengg-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: false - key: prd/admin/coralogix-keys - secretStoreRef: - name: vault-backend - -image: - # If you want to use the core image `otel/opentelemetry-collector`, you also need to change `command.name` value to `otelcol`. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - # repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - tag: "0.153.0" - # When digest is set to a non-empty value, images will be pulled by digest (regardless of tag value). - digest: "" -mode: daemonset -rollout: - rollingUpdate: - # When 'mode: daemonset', maxSurge cannot be used when hostPort is set for any of the ports - # maxSurge: 25% - maxUnavailable: 1 - strategy: RollingUpdate -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" -hostNetwork: true -dnsPolicy: "ClusterFirstWithHostNet" -fullnameOverride: opentelemetry-dataengg-prd - -presets: - logsCollection: - enabled: false - storeCheckpoints: false - maxRecombineLogSize: 1048576 - extraFilelogOperators: [] -# - type: recombine -# combine_field: body -# source_identifier: attributes["log.file.path"] -# is_first_entry: body matches "^(YOUR-LOGS-REGEX)"s - kubernetesAttributes: - enabled: false - hostMetrics: - enabled: false - kubeletMetrics: - enabled: false - -extraEnvs: -- name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" -- name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName -config: - extensions: {} - # zpages: - # endpoint: localhost:55679 - # pprof: - # endpoint: localhost:1777 - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 4s - tls: - insecure: true - sending_queue: - enabled: true - queue_size: 6000 - num_consumers: 200 - resolver: - dns: - hostname: opentelemetry-admin-prd.opentelemetry.svc.clusterset.local - # coralogix: - # timeout: "30s" - # private_key: "${CORALOGIX_PRIVATE_KEY}" - # domain: "{{.Values.global.domain}}" - # traces: - # endpoint: "{{ .Values.global.traces.endpoint }}" - # metrics: - # endpoint: "{{ .Values.global.metrics.endpoint }}" - # logs: - # endpoint: "{{ .Values.global.logs.endpoint }}" - # application_name_attributes: - # - "k8s.namespace.name" - # - "service.namespace" - # subsystem_name_attributes: - # - "k8s.deployment.name" - # - "k8s.statefulset.name" - # - "k8s.daemonset.name" - # - "k8s.cronjob.name" - # - "k8s.job.name" - # - "k8s.container.name" - # - "k8s.node.name" - # - "service.name" - # application_name: "{{.Values.global.defaultApplicationName }}" - # subsystem_name: "{{.Values.global.defaultSubsystemName }}" - processors: - k8s_attributes: - filter: - node_from_env_var: KUBE_NODE_NAME - extract: - metadata: - - "k8s.namespace.name" - - "k8s.deployment.name" - - "k8s.statefulset.name" - - "k8s.daemonset.name" - - "k8s.cronjob.name" - - "k8s.job.name" - - "k8s.pod.name" - - "k8s.node.name" - memory_limiter: null # Will get the k8s resource limits - resourcedetection/env: - detectors: ["system","env"] - timeout: 2s - override: false - # spanmetrics: - # metrics_exporter: coralogix - # dimensions: - # - name: "k8s.deployment.name" - # - name: "k8s.statefulset.name" - # - name: "k8s.daemonset.name" - # - name: "k8s.cronjob.name" - # - name: "k8s.job.name" - # - name: "k8s.container.name" - # - name: "k8s.node.name" - # - name: "k8s.namespace.name" - receivers: - otlp: - protocols: - grpc: - endpoint: ${MY_POD_IP}:4317 - http: - endpoint: ${MY_POD_IP}:4318 - zipkin: - endpoint: ${MY_POD_IP}:9411 - jaeger: - protocols: - grpc: - endpoint: ${MY_POD_IP}:14250 - thrift_http: - endpoint: ${MY_POD_IP}:14268 - thrift_compact: - endpoint: ${MY_POD_IP}:6831 - thrift_binary: - endpoint: ${MY_POD_IP}:6832 - prometheus: - config: - scrape_configs: - - job_name: opentelemetry-collector - scrape_interval: 30s - static_configs: - - targets: - - ${MY_POD_IP}:8888 - service: - extensions: - # - zpages - # - pprof - - health_check -# - memory_ballast - telemetry: - metrics: - address: ${env:MY_POD_IP}:8888 - pipelines: - traces: - exporters: - - loadbalancing - processors: - - memory_limiter - # - spanmetrics - - batch - receivers: - - otlp - metrics: null - logs: null -tolerations: - - operator: Exists - -resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 1 - memory: 2G - -ports: - jaeger-binary: - enabled: true - containerPort: 6832 - servicePort: 6832 - hostPort: 6832 - protocol: TCP - # In order to enable podMonitor, following part must be enabled in order to expose the required port: - # metrics: - # enabled: true - -# podMonitor: -# enabled: true - -# prometheusRule: -# enabled: true -# defaultRules: -# enabled: true - -# Annotations to be added to pod -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-dataengg-contour-ext - operator: NotIn - values: - - dedicated - - key: p-dataengg-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-dataengg-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-dataengg-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-dataengg-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert \ No newline at end of file diff --git a/helm-overrides/gke-dataengg-prd-ase1a/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index 0fabfc1..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,134 +0,0 @@ -fullnameOverride: opentelemetry-dataengg-prd - -mode: daemonset - -priorityClassName: "system-node-critical" - -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.153.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-dataengg-contour-ext - operator: NotIn - values: - - dedicated - - key: p-dataengg-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-dataengg-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-dataengg-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-dataengg-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external - -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 100m - memory: 124Mi - limits: - cpu: 200m - memory: 256Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "dataengg" - team: "sre" - service: "opentelemetry-dataengg-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/gke-dataengg-prd-ase1a/prometheus-node-exporter/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index 2c91b73..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,487 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-dataengg-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "node-exporter-dataengg-a-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/gke-dataengg-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index 1d997ce..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,169 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-dataengg-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-dataengg-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "stackdriver-exporter-dataengg-a-prd" - env: "prd" - priority: "p0" - type: "exporter" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-dataengg-prd-0622" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'compute.googleapis.com/instance,cloudsql.googleapis.com/database,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - cloud.google.com/compute-class: dataengg-devops - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - effect: "NoSchedule" - key: "cloud.google.com/compute-class" - value: "dataengg-devops" - operator: "Equal" - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-deng-desre-stackdriver-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - } - - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/gke-dataengg-prd-ase1a/telegraf-operator/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/telegraf-operator/custom-values.yaml deleted file mode 100644 index 8b4e228..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,151 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "dataengg" - team: "dataengg-sre" - service: "telegraf-operator-dataengg-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - cloud.google.com/compute-class: "dataengg-devops" - -tolerations: - - effect: "NoSchedule" - key: "cloud.google.com/compute-class" - value: "dataengg-devops" - operator: "Equal" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false diff --git a/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml deleted file mode 100644 index 31b8c01..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml +++ /dev/null @@ -1,296 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-dataengg-prd-dr -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-dataengg-prd-dr-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent-dr - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-deng-desre-vmagent-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - - https://vminsert-infra-prd-dr.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vmagent-dataengg-prd" - env: "prd" - priority: "p0" - type: "vmagent-dr" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dataengg-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dataengg-prd-dr.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 4 - memory: 6Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - cloud.google.com/compute-class: "vmagent-dr" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml deleted file mode 100644 index 89655d9..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml +++ /dev/null @@ -1,272 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 1 - -fullnameOverride: vmagent-dataengg-prd-fb -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-dataengg-prd-fb-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - -# Remove topology spread constraints -topologySpreadConstraints: [] - -# Modify deployment strategy -deployment: - enabled: true - -# Remove HPA -horizontalPodAutoscaler: - enabled: false - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-infr-sre-vmagent-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -# Remove pod disruption budget -podDisruptionBudget: - enabled: false - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-dataengg.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vmsinglenode-infra-prd.victoriametrics.svc.clusterset.local:8428/api/v1/write - - https://vmsinglenode-infra-prd.meeshogcp.in/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dataengg-fb" - team: "sre" - service: "vmagent-dataengg-a-prd-fb" - env: "prd" - priority: "p0" - type: "vmagent-fb" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dataengg-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dataengg-a-prd-fb.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -# Modify resources -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - requests: - cpu: 2 - memory: 2Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - cloud.google.com/compute-class: "vmagent-mds" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-mds" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml deleted file mode 100644 index 3005ffb..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert-dr - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-dataengg-prd-dr - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/dataengg/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-dataengg-prd-dr.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-dataengg-prd-dr" - env: "prd" - priority: "p0" - type: "vmalert-dr" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmagent-dr" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-alert-startree-stateful/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-alert-startree-stateful/custom-values.yaml deleted file mode 100644 index dc975fb..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-alert-startree-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-dataengg-startree-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-dataengg-startree-prd-proxy.victoriametrics-startree.svc.cluster.local:8481/select/0/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-dataengg-startree-prd.victoriametrics-startree.svc.cluster.local:8480/insert/0/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-dataengg-startree-prd-proxy.victoriametrics-startree.svc.cluster.local:8481/select/0/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/pinot/data-intelligence/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-dataengg-startree-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-dataengg-startree-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmstack-startree" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmstack-startree" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml deleted file mode 100644 index 2e716c2..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-dataengg-a-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-dataengg-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-dataengg-a-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-dataengg-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/dataengg/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-dataengg-a-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "dataengg" - team: "sre" - service: "vmalert-dataengg-a-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-insert-startree/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-insert-startree/custom-values.yaml deleted file mode 100644 index bef609b..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-insert-startree/custom-values.yaml +++ /dev/null @@ -1,234 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dataengg-startree-prd - replicaCount: 3 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-dataengg-startree-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 50 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vminsert-dataengg-startree-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 2 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 60 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmstack-startree" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - cloud.google.com/compute-class: "vmstack-startree" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 2 - memory: 1.6Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-dataengg-startree-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: nginx-internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-select-startree/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-select-startree/custom-values.yaml deleted file mode 100644 index 76d3193..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-select-startree/custom-values.yaml +++ /dev/null @@ -1,293 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dataengg-startree-prd - replicaCount: 3 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-dataengg-startree-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 180s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 96 - clusternative.maxConcurrentRequests: 256 - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vmselect-dataengg-startree-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 25 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - cloud.google.com/compute-class: "vmstack-startree" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmstack-startree" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 5 - memory: 18Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-dataengg-startree-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-storage-startree/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-storage-startree/custom-values.yaml deleted file mode 100644 index 98741e4..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/victoria-metrics-storage-startree/custom-values.yaml +++ /dev/null @@ -1,316 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-dataengg-startree-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmstack-startree" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - cloud.google.com/compute-class: "vmstack-startree" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 500Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vmstorage-dataengg-startree-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 3 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 6 - memory: 16Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.93.7-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-dataengg-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/gke-dataengg-prd-ase1a/victoriametrics-agent/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index ed8664d..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,488 +0,0 @@ -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "dataengg-a-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -# -- Replica count -replicaCount: 2 - -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - -# -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables -deployment: - spec: - minReadySeconds: 1200 - progressDeadlineSeconds: 1800 - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - -image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - # -- Image tag, set to `Chart.AppVersion` by default - tag: v1.133.0 # rewrites Chart.AppVersion - # -- Variant of the image to use. - # e.g. enterprise, scratch - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - -# -- Image pull secrets -imagePullSecrets: [] - -# -- Add additional DNS entries to pods hosts file. Check [official documentation](https://kubernetes.io/docs/tasks/network/customize-hosts-file-for-pods/) -hostAliases: [] -# - ip: 192.168.1.1 -# hostNames: -# - test.example.com -# - another.example.net - -# -- Override chart name -nameOverride: "" - -# -- Override resources fullname -fullnameOverride: "vm-agent-dataengg-a-prd" - -# -- Container working directory -containerWorkingDir: "/" - -rbac: - # -- Enables Role/RoleBinding creation - create: true - - # -- Role/RoleBinding annotations - annotations: {} - - # -- Role/RoleBinding labels - extraLabels: {} - - # -- If true and `rbac.enabled`, will deploy a Role/RoleBinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - - # -- additional rules for a role - extraRules: [] - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - # -- Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-deng-sre-vmagnt-prd-mds@meesho-dataengg-prd-0622.iam.gserviceaccount.com - } - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # -- mount API token to pod directly - automountToken: true - -# -- See `kubectl explain poddisruptionbudget.spec` for more or check [official documentation](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# -- Generates `remoteWrite.*` flags and config maps with value content for values, that are of type list of map. -# Each item should contain `url` param to pass validation. -remoteWrite: - - url: http://vm-insert-dataengg-a-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write - disableOnDiskQueue: false - dropSamplesOnOverload: false - # - url: http://vm-insert-dataengg-prd.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/api/v1/write - # disableOnDiskQueue: false - # dropSamplesOnOverload: false - - url: http://prd-census-server-dataengg.prd-census-server-dataengg.svc.cluster.local/api/v1/write - disableOnDiskQueue: true - dropSamplesOnOverload: true -# urlRelabelConfig: -# - action: keep -# source_labels: [env] -# regex: "dev" - -# -- VMAgent extra command line arguments -extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8429 - promscrape.config.strictParse: false # default is true - promscrape.maxScrapeSize: 1000000000 # default is 167772160 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - # promscrape.suppressScrapeErrors: true # default is false - - # Uncomment and specify the port if you want to support any of the protocols: - # https://docs.victoriametrics.com/victoriametrics/vmagent/#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for more details. -env: - - name: GOGC - value: "200" - - name: DRAGONFLY_BEARER_TOKEN - valueFrom: - secretKeyRef: - key: token - name: dragonfly-cloud-creds - -# -- Specify alternative source for env variables -envFrom: - [] - #- configMapRef: - # name: special-config - -# -- Extra labels for Deployment and Statefulset -extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vm-agent-dataengg-a-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra labels for Pods only -podLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vm-agent-dataengg-a-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra selector labels common for pod and service -selectorLabels: {} - -# -- Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# -- Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# -- Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -# -- Extra containers to run in a pod with vmagent -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -# -- Init containers for vmagent -initContainers: - [] - # - name: example - # image: example-image - -# -- Security context to be added to pod -podSecurityContext: - enabled: true - # fsGroup: 2000 - -# -- Security context to be added to pod's containers -securityContext: - enabled: true - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - # -- Enable agent service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - extraLabels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) for details - externalIPs: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8429 - # -- Target port - targetPort: http - # nodePort: 30000 - # -- Service type - type: ClusterIP - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - -ingress: - # -- Enable deployment of ingress for agent - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-agent-dataengg-a-prd.meeshogcp.in - path: - - / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - - # -- Ingress controller class name - ingressClassName: "nginx-internal" - - # -- Ingress path type - pathType: Prefix - -# -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) -resources: - requests: - cpu: 25 - memory: 24Gi - -# -- Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) -nodeSelector: - cloud.google.com/compute-class: "vmagent-mds" - -# -- Node tolerations for server scheduling to nodes with taints. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-mds" - effect: "NoSchedule" - -# -- Pod topologySpreadConstraints -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# -- Pod affinity -affinity: {} - -# -- VMAgent [scraping configuration](https://docs.victoriametrics.com/victoriametrics/vmagent/#how-to-collect-metrics-in-prometheus-format) -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-dataengg-a-prd" - -# -- Priority class to be assigned to the pod(s) -priorityClassName: "" - -# -- Enable the host network -hostNetwork: false - -serviceMonitor: - # -- Enable deployment of Service Monitor for server component. This is Prometheus operator object - enabled: false - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Service Monitor relabelings - relabelings: [] - # -- Basic auth params for Service Monitor - basicAuth: {} - # -- Service Monitor metricRelabelings - metricRelabelings: [] - # -- Service Monitor targetPort - targetPort: http - # interval: 15s - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - -# -- Empty dir configuration for a case, when persistence is disabled -emptyDir: {} - -persistentVolume: - # -- Create/use Persistent Volume Claim for server component. Empty dir if false - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Size of the volume. Should be calculated based on the logs you send and retention policy you set. - size: 10Gi - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume additional labels - extraLabels: {} - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Horizontal Pod Autoscaling. -# Note that it is not intended to be used for vmagents which perform scraping. -# In order to scale scraping vmagents check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) -horizontalPodAutoscaling: - # -- Use HPA for vmagent - enabled: false - # -- Maximum replicas for HPA to use to to scale vmagent - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale vmagent - minReplicas: 1 - # -- Metric for HPA to use to scale vmagent - metrics: [] - -# -- VMAgent scrape configuration -config: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -probe: - # -- Readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - # -- Liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - # -- Startup probe - startup: {} - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -allowedMetricsEndpoints: - - /metrics - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-dataengg-prd-ase1a/victoriametrics-insert/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/victoriametrics-insert/custom-values.yaml deleted file mode 100644 index b39b614..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/victoriametrics-insert/custom-values.yaml +++ /dev/null @@ -1,411 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: false - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- IDs of vmstorage nodes to exclude from writing - excludeStorageIDs: [] - # -- Override default `app` label name - name: vminsert - # -- VMInsert strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - # override Chart.AppVersion - tag: v1.107.0-cluster - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMInsert http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-dataengg-a-prd - # -- Extra command line arguments for vminsert component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8480 - maxLabelsPerTimeseries: 55 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-dataengg-a-prd-0.vm-storage-dataengg-a-prd.victoriametrics.svc:8400" - - "vm-storage-dataengg-a-prd-1.vm-storage-dataengg-a-prd.victoriametrics.svc:8400" - - "vm-storage-dataengg-a-prd-2.vm-storage-dataengg-a-prd.victoriametrics.svc:8400" - - "vm-storage-dataengg-a-prd-3.vm-storage-dataengg-a-prd.victoriametrics.svc:8400" - - "vm-storage-dataengg-a-prd-4.vm-storage-dataengg-a-prd.victoriametrics.svc:8400" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vm-insert-dataengg-a-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Pod's additional labels - podLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vm-insert-dataengg-a-prd" - env: "prd" - priority: "p0" - type: "vminsert" - terminationGracePeriodSeconds: 30 - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - - # -- Readiness & Liveness probes - probe: - # -- VMInsert readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert startup probe - startup: {} - - # -- Relabel configuration - relabel: - enabled: false - config: [] - # -- Use existing configmap if specified - # otherwise .config values will be used. Relabel config **should** reside under `relabel.yml` key - configMap: "" - - # Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 4 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 60 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vminsert - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vminsert - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - cloud.google.com/compute-class: "vminsert-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 10 - memory: 10Gi - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - service: - # -- Create VMInsert service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here]( https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) for details - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-insert-dataengg-a-prd.meeshogcp.in - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - - # -- Ingress controller class name - ingressClassName: nginx-internal - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for insert component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/insert" .Values.vminsert }}' - - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vminsert component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-dataengg-prd-ase1a/victoriametrics-select/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/victoriametrics-select/custom-values.yaml deleted file mode 100644 index 2390e05..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/victoriametrics-select/custom-values.yaml +++ /dev/null @@ -1,447 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - splitService: true - externalService: - enabled: true - name: vmselect-dataengg-a-prd-proxy - # -- Override default `app` label name - clusternativeService: - enabled: false - name: "" - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # override Chart.AppVersion - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMSelect http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-dataengg-a-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - # -- Extra command line arguments for vmselect component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8481 - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 180s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-dataengg-a-prd-0.vm-storage-dataengg-a-prd.victoriametrics.svc:8401" - - "vm-storage-dataengg-a-prd-1.vm-storage-dataengg-a-prd.victoriametrics.svc:8401" - - "vm-storage-dataengg-a-prd-2.vm-storage-dataengg-a-prd.victoriametrics.svc:8401" - - "vm-storage-dataengg-a-prd-3.vm-storage-dataengg-a-prd.victoriametrics.svc:8401" - - "vm-storage-dataengg-a-prd-4.vm-storage-dataengg-a-prd.victoriametrics.svc:8401" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vm-select-dataengg-a-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # -- Pod's additional labels - podLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vm-select-dataengg-a-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Readiness & Liveness probes - probe: - # -- VMSelect readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 25 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 5 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vmselect - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vmselect - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 17 - memory: 31Gi - - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Create VMSelect service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-select-dataengg-a.prd.meesho.int - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # -- Ingress controller class name - ingressClassName: contour-internal-0 - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for select component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/select" .Values.vmselect }}' - - # -- vmselect mode: deployment, daemonSet - mode: deployment - - # -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables - deployment: - spec: - # -- VMSelect strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - - # -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables - statefulSet: - enabled: false - spec: - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume extra labels - extraLabels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmselect component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: false - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - diff --git a/helm-overrides/gke-dataengg-prd-ase1a/victoriametrics-storage/custom-values.yaml b/helm-overrides/gke-dataengg-prd-ase1a/victoriametrics-storage/custom-values.yaml deleted file mode 100644 index 70c821e..0000000 --- a/helm-overrides/gke-dataengg-prd-ase1a/victoriametrics-storage/custom-values.yaml +++ /dev/null @@ -1,363 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - enabled: false - -vminsert: - enabled: false - -vmauth: - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- Override default `app` label name - name: vmstorage - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Variant of the image to use. e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMStorage http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-dataengg-a-prd - - # -- Additional environment variables (ex.: secret tokens, flags). Check https://docs.victoriametrics.com/victoriametrics/#environment-variables for details - env: [] - # -- Specify alternative source for env variables - envFrom: [] - # -- Data retention period. Possible units character: h(ours), d(ays), w(eeks), y(ears), if no unit character specified - month. The minimum retention period is 24h. - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - httpListenAddr: :8482 - loggerTimezone: "Asia/Kolkata" - dedup.minScrapeInterval: 60s - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmstorage-n4d" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - cloud.google.com/compute-class: "vmstorage-n4d" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - # -- Use an alternate scheduler, e.g. "stork". Check https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ for details - # schedulerName: - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Override Persistent Volume Claim name - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are https://kubernetes.io/docs/concepts/storage/persistent-volumes/ - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume extra labels - extraLabels: {} - # -- Storage class name. Will be empty if not set - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume - size: 1300Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vm-storage-dataengg-a-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - # -- Pod's additional labels - podLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vm-storage-dataengg-a-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Details are https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 26 - memory: 220Gi - - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: - enabled: false - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - podSecurityContext: - enabled: false - service: - enabled: true - # -- Service annotations - annotations: {} - # -- Service ClusterIP - clusterIP: None - # -- Service type - type: ClusterIP - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmselect - vmselectPort: 8401 - # -- Extra service ports - extraPorts: [] - # -- Health check node port for a service. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilyPolicy: "" - # -- List of service IP families. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilies: [] - # -- Traffic Distribution. Check https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution - trafficDistribution: "" - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - minReadySeconds: 5 - # -- Readiness probes - probe: - # -- VMStorage readiness probe - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMStorage startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmstorage component - enabled: false - # -- Maximum replicas for HPA to use to to scale the vmstorage component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmstorage component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmstorage component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: - scaleDown: - selectPolicy: Disabled - - vmbackupmanager: - # -- Enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" \ No newline at end of file diff --git a/helm-overrides/gke-datascience-prd-as1a/README.md b/helm-overrides/gke-datascience-prd-as1a/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/gke-datascience-prd-as1a/alloy/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/alloy/custom-values.yaml deleted file mode 100644 index d2a053e..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/alloy/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -fullnameOverride: "alloy-datascience-prd" - -alloy: - configMap: - configFile: datascience.alloy - - clustering: - enabled: true - - extraPorts: - - name: "otlp-grpc" - port: 4317 - targetPort: 4317 - protocol: "TCP" - - name: "otlp-http" - port: 4318 - targetPort: 4318 - protocol: "TCP" - - resources: - requests: - cpu: 7 - memory: 55Gi - -configReloader: - enabled: true - -serviceAccount: - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-sre-grafna-obs-stk-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - -controller: - # Must be one of 'daemonset', 'deployment', or 'statefulset'. - type: 'deployment' - - nodeSelector: - cloud.google.com/compute-class: "alloy" - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "alloy" - effect: "NoSchedule" - - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 50 - targetCPUUtilizationPercentage: 80 - targetMemoryUtilizationPercentage: 80 \ No newline at end of file diff --git a/helm-overrides/gke-datascience-prd-as1a/aurva-dataplane/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/aurva-dataplane/custom-values.yaml deleted file mode 100644 index 3ccab83..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/aurva-dataplane/custom-values.yaml +++ /dev/null @@ -1,674 +0,0 @@ -## Aurva Data Plane -## Ref: https://github.com/aurva-io/aurva-charts.git - -postgresql: - enabled: true - fullnameOverride: "aurva-dataplane-database" - volumePermissions: - ## @param volumePermissions.enabled Enable init container that changes the owner and group of the persistent volume - ## - enabled: true - global: - storageClass: pd-standard-retain - postgresql: - auth: - postgresPassword: "aurva" - database: "controller" - # Add toleration to make sure where this postgres db pod should reside (Applicable for production workloads): For more detail ref: https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/ - primary: - extendedConfiguration: | - max_connections = 300 - - # #PLACEHOLDER## - # tolerations: - # - effect: NoSchedule - # key: cloud.google.com/compute-class - # operator: Equal - # value: datascience-devops - - # # -- Select nodes to deploy which matches the following labels - # nodeSelector: ##PLACEHOLDER## - # cloud.google.com/compute-class: datascience-devops - -# -- Provide a name in place of `aurva` -# namespaceOverride: aurva-dataplane - -########################################################## -# Global Configs -########################################################## -global: - aurva_controller: - enabled: true - aurva_fastdet: - enabled: false - aurva_pii_analyzer: - enabled: true - aurva_ocr: - enabled: true - aurva_collector: - enabled: true - - deploymentAnnotations: {} - - priorityClassName: "" - -########################################################## -# Aurva Controller -########################################################## -aurva_controller: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "datascience" - team: "datascience-devops" - service: "aurva-datascience-prd" - env: "prd" - priority: "p0" - type: "aurva_controller" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 10 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - image: - # Image of the app container - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-controller/aurva-controller - tag: "v3.18.2" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - - type: secret - name: aurva-controller-secrets - # - type: configmap - # name: proxy-datasource-config - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 2 - requests: - memory: 1Gi - cpu: 1 - - aurvaFastdet: - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-fastdet/aurva-fastdet - tag: "v2.30.13" - pullPolicy: IfNotPresent - envFrom: [] - env: [] - resources: - limits: - cpu: 0.5 - memory: 512Mi - requests: - cpu: 0.5 - memory: 512Mi - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: datascience-devops - - # -- Taint tolerations for nodes - ##PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-controller-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - config: - #variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - COMMAND_URL: "command.aurva-prd.meeshogcp.in:80" - DEPLOYMENT_TYPE: "kubernetes" - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - FLUSHER_WORKER_POOL_SIZE: "1000" - FLUSHER_BATCH_SIZE: "30000" - UNIQUENESS_IDENTIFIER: "k8s-datascience-prd-ase1" #Recommendation: should be equal to cluster name - PROVIDER_ACCOUNT_ID: "meesho-datascience-prd-0622" # GCP PROJECT ID (not Number) - REGION: "asia-southeast1" #eg: asia-south1 - ENVIRONMENT: "prod" - OCR_ENABLED: "true" - MONITORING_ENABLED: "false" - HYBRID_ONLY_MODE: "false" - FORCE_TLS: "false" - FASTDET_FLAG : "true" - AADHAAR_ENHANCER: "0" - WORKSPACE_EVENT_TRACKING_ENABLED: "false" - ACCESS_IQ_ENABLED: "true" - GRPC_ENFORCE_ALPN_ENABLED: "false" - ENABLE_FASTDET: "true" - FASTDET_MAX_BATCH_SIZE: "100" - HEARTBEAT_INTERVAL: "5m" - #pii data - PII_BUCKET_NAME: "gcs-infra-devop-aurva-datascience-prd" - PII_LOG_BUCKET_REGION: "asia-southeast1" - PII_LOG_CRON: "*/5 * * * *" - ENABLE_PII_LOG: "true" - PII_EVIDENCE_UPLOAD_MAX_BLOCKING_TASKS: "50000" - PII_EVIDENCE_UPLOAD_MAX_CONCURRENT_TASKS: "500" - PII_EVIDENCE_MAX_CACHE_WEIGHT: "100" - QUOTA_CLEANUP_CRON: "0 0 * * *" - ENABLE_PII_QUOTA: "true" - MAX_PII_EVIDENCES_PER_KEY_PER_WINDOW: "3" - PII_QUOTA_SYNC_CRON: "*/2 * * * *" - QUOTA_WINDOW_HOURS: "24" - #constants - SKIP_NAMESPACES: "contour-internal-1-datascience-prd,contour-internal-0-datascience-prd,contour-external-datascience-prd" - CLOUD_PROVIDER: "gcp" - LOG_ENV: "production" - RDS_SCANNER_AVAILABILITY : "false" - REDSHIFT_SCANNER_AVAILABILITY : "false" - S3_SCANNER_AVAILABILITY : "false" - DYNAMO_SCANNER_AVAILABILITY: "false" - DOCDB_SCANNER_AVAILABILITY: "false" - OPENSEARCH_SCANNER_AVAILABILITY: "false" - CLOUDSQL_SCANNER_AVAILABILITY: "true" - BIGQUERY_SCANNER_AVAILABILITY: "true" - AWS_SNAPSHOT_SCANNER_AVAILABILITY: "false" - CLOUDSTORAGE_SCANNER_AVAILABILITY: "true" - KEYSPACES_SCANNER_AVAILABILITY: "false" - ALLOYDB_SCANNER_AVAILABILITY: "true" - BIGTABLE_SCANNER_AVAILABILITY: "true" - GCP_BACKUP_AVAILABILITY: "true" - EGRESS_MODE_ONLY: "false" - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-controller-sa - # -- Annotations applied to created service account - annotations: - iam.gke.io/gcp-service-account: sa-datascience-prd-aurva-contr@meesho-datascience-prd-0622.iam.gserviceaccount.com -# eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 15 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 - - -########################################################## -# Aurva OCR -########################################################## -aurva_ocr: - # -- Additional labels for aurva-controller - additionalLabels: - bu: "datascience" - team: "datascience-devops" - service: "aurva-datascience-prd" - env: "prd" - priority: "p0" - type: "aurva_ocr" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 1 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-ocr/aurva-ocr - tag: "v3.15.11" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-ocr: - type: secret - name: aurva-ocr-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 1 - requests: - memory: 2Gi - cpu: 1 - - # -- Select nodes to deploy which matches the following labels - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: datascience-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-ocr-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - OCR_TIME_LIMIT: "1" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-datascience-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-ocr-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # iam.gke.io/gcp-service-account: service-account@gcp.iam.gserviceaccount.com - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva Collector -########################################################## -aurva_collector: - - # -- Additional labels for aurva-analyzer - additionalLabels: - bu: "datascience" - team: "datascience-devops" - service: "aurva-datascience-prd" - env: "prd" - priority: "p0" - type: "aurva_collector" - - - # -- Annotations on aurva-analyzer - annotations: {} -# "key": "value" - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-collector/aurva-collector - tag: "v3.18.2" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-controller: - type: secret - name: aurva-collector-secrets - - resources: - limits: - cpu: 800m - memory: 800Mi - requests: - cpu: 200m - memory: 512Mi - - podSecurityContext: {} - - securityContext: - privileged: true - capabilities: - add: - # For kernel v5.8 and above we don't need CAP_SYS_ADMIN or CAP_SYS_RESOURCE - # we just need CAP_BPF and CAP_PERFMON. This has been tested on our EKS node - # which is on kernel v5.10.x - # When SSL Tracing is required we need CAP_SYS_ADMIN and CAP_SYS_PTRACE - # on top of the previous capabilities - # So finally these are the 4 possible combinations for capabilies - # 1. Newer Kernels without SSL - # - BPF - # - PERFMON - # 2. Newer Kernels with SSL - - SYS_ADMIN - - SYS_PTRACE - # 3. Older Kernels without SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # 4. Older Kernels with SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # - SYS_PTRACE - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - volumes: - - name: debugfs - mountPath: /sys/kernel/debug - hostPath: /sys/kernel/debug - - name: vmlinux - mountPath: /sys/kernel/btf/vmlinux - hostPath: /sys/kernel/btf/vmlinux - - name: procfs - mountPath: /host/proc - hostPath: /proc - - name: bpffs - mountPath: /sys/fs/bpf - hostPath: /sys/fs/bpf - - # -- Taint tolerations for nodes - tolerations: - # - effect: NoSchedule - # key: cloud.google.com/compute-class - # operator: Equal - # value: megatetra - - operator: Exists - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: cloud.google.com/compute-class - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external - - alloy - - preprod-spot-16 - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-collector-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - # variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-datascience-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - TRACE_INTERNAL_SVC: "true" - TRACE_INTERNAL_SVC_HTTP: "true" - INTERNAL_SVC_SAMPLE_INTERVAL: "10m" - LOGS_TTL: "1h" - TRACE_HTTP2: "true" - TRACE_SSL: "false" - TRACE_PSQL: "false" - TRACE_SQLSERVER: "false" - TRACE_MYSQL: "false" - TRACE_EGRESS: "true" - TRACE_GO_TLS: "false" - TRACE_ML_SERVICES: "false" - # constants - LOG_ENV: production - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - MONITORING_ENABLED: "false" - ENABLE_INGRESS_INFORMER: "false" - ENABLE_SERVICE_INFORMER: "false" - ENABLE_ISTIO_INFORMER: "false" - EXCLUDED_PII_REGEX_TYPES: "ip_address,us_bank_number,us_driver_license,us_itin,us_passport,us_routing,us_mbi,ssn" - AGGREGATOR_MAX_CONNECTIONS: "1000" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-collector-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva PII Analyzer -########################################################## -aurva_pii_analyzer: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "datascience" - team: "datascience-devops" - service: "aurva-datascience-prd" - env: "prd" - priority: "p0" - type: "aurva_pii_analyzer" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 3 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - ##PLACEHOLDER## - nodeSelector: - cloud.google.com/compute-class: datascience-devops - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-piianalyzer/aurva-piianalyzer - tag: "v3.15.11" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-pii-analyzer: - type: secret - name: aurva-pii-analyzer-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 4Gi - cpu: 4 - requests: - memory: 2Gi - cpu: 2 - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: datascience-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-pii-analyzer-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - SCHEDULER_TIME: "1" - SUPPORTED_REGION: "US" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-datascience-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-pii-analyzer-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 3 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 diff --git a/helm-overrides/gke-datascience-prd-as1a/cert-manager/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/cert-manager/custom-values.yaml deleted file mode 100644 index 6ee52f1..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/cert-manager/custom-values.yaml +++ /dev/null @@ -1,129 +0,0 @@ -global: - logLevel: 2 - rbac: - create: true - priorityClassName: "high-priority" -installCRDs: false - -crds: - enabled: true - keep: true - -# Cert-manager Controller -replicaCount: 1 -image: - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-controller - tag: v1.20.1 - pullPolicy: IfNotPresent - -nodeSelector: - cloud.google.com/compute-class: datascience-devops - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - -resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 500m - memory: 512Mi - -# Webhook Configuration -webhook: - replicaCount: 1 - image: - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-webhook - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: datascience-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# CA Injector Configuration -cainjector: - enabled: true - replicaCount: 1 - image: - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-cainjector - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: datascience-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# ACME Solver Configuration -acmesolver: - image: - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-acmesolver - tag: v1.20.1 - pullPolicy: IfNotPresent - -# Startup API Check -startupapicheck: - enabled: true - timeout: 1m - backoffLimit: 4 - image: - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-startupapicheck - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: datascience-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - -# Prometheus Monitoring -prometheus: - enabled: true - servicemonitor: - enabled: false - interval: 60s - scrapeTimeout: 30s - labels: - prometheus: cert-manager diff --git a/helm-overrides/gke-datascience-prd-as1a/computeclass/c3-highcpu-44-compute-class.yaml b/helm-overrides/gke-datascience-prd-as1a/computeclass/c3-highcpu-44-compute-class.yaml deleted file mode 100644 index cb05103..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/computeclass/c3-highcpu-44-compute-class.yaml +++ /dev/null @@ -1,23 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: c3-highcpu-44-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - labels: - dedicated: megaduo - serviceAccount: sa-dsci-shared-nodepool-prd@meesho-dsci-as1-prd-0622.iam.gserviceaccount.com - priorities: - - location: - zones: - - asia-south1-a - machineType: c3-highcpu-44 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-as1a/computeclass/c3d-highcpu-30-compute-class.yaml b/helm-overrides/gke-datascience-prd-as1a/computeclass/c3d-highcpu-30-compute-class.yaml deleted file mode 100644 index 6b62f6f..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/computeclass/c3d-highcpu-30-compute-class.yaml +++ /dev/null @@ -1,23 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: c3d-highcpu-30-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - labels: - dedicated: mlp-c3d-highcpu-30-v1 - serviceAccount: sa-dsci-shared-nodepool-prd@meesho-dsci-as1-prd-0622.iam.gserviceaccount.com - priorities: - - location: - zones: - - asia-south1-a - machineType: c3d-highcpu-30 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-as1a/computeclass/c4a-highcpu-16-compute-class.yaml b/helm-overrides/gke-datascience-prd-as1a/computeclass/c4a-highcpu-16-compute-class.yaml deleted file mode 100644 index 4a3fe44..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/computeclass/c4a-highcpu-16-compute-class.yaml +++ /dev/null @@ -1,23 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: c4a-highcpu-16-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - labels: - dedicated: rust-onboard - serviceAccount: sa-dsci-shared-nodepool-prd@meesho-dsci-as1-prd-0622.iam.gserviceaccount.com - priorities: - - location: - zones: - - asia-south1-a - machineType: c4a-highcpu-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-0-arm.yaml b/helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-0-arm.yaml deleted file mode 100644 index ef935bd..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-0-arm.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-0-arm -spec: - nodePoolConfig: - serviceAccount: sa-dsci-shared-nodepool-prd@meesho-dsci-as1-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-south1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-1-arm.yaml b/helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-1-arm.yaml deleted file mode 100644 index 56413ca..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-1-arm.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-1-arm -spec: - nodePoolConfig: - serviceAccount: sa-dsci-shared-nodepool-prd@meesho-dsci-as1-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-south1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-dataproc-arm.yaml b/helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-dataproc-arm.yaml deleted file mode 100644 index 6cdc01d..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-dataproc-arm.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-dataproc-arm -spec: - nodePoolConfig: - serviceAccount: sa-dsci-shared-nodepool-prd@meesho-dsci-as1-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-south1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-intra-0-arm.yaml b/helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-intra-0-arm.yaml deleted file mode 100644 index 615e671..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-intra-0-arm.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-intra-0-arm -spec: - nodePoolConfig: - serviceAccount: sa-dsci-shared-nodepool-prd@meesho-dsci-as1-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-south1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-intra-1-arm.yaml b/helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-intra-1-arm.yaml deleted file mode 100644 index dadce26..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/computeclass/contour-internal-intra-1-arm.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-intra-1-arm -spec: - nodePoolConfig: - serviceAccount: sa-dsci-shared-nodepool-prd@meesho-dsci-as1-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-south1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-32 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-32 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-as1a/computeclass/contour-shared-arm.yaml b/helm-overrides/gke-datascience-prd-as1a/computeclass/contour-shared-arm.yaml deleted file mode 100644 index 5598984..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/computeclass/contour-shared-arm.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-shared-arm -spec: - nodePoolConfig: - serviceAccount: sa-dsci-shared-nodepool-prd@meesho-dsci-as1-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-south1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-as1a/computeclass/datascience-devops.yaml b/helm-overrides/gke-datascience-prd-as1a/computeclass/datascience-devops.yaml deleted file mode 100644 index f1e0735..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/computeclass/datascience-devops.yaml +++ /dev/null @@ -1,34 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: datascience-devops -spec: - nodePoolConfig: - serviceAccount: sa-dsci-shared-nodepool-prd@meesho-dsci-as1-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-south1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-standard-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 60 - bootDiskType: hyperdisk-balanced - - machineType: c4-standard-4 - spot: false - maxPodsPerNode: 32 - storage: - bootDiskSize: 60 - bootDiskType: hyperdisk-balanced - - machineType: e2-standard-4 - spot: false - maxPodsPerNode: 32 - storage: - bootDiskSize: 60 - bootDiskType: pd-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-as1a/computeclass/g2-standard-16-l4-compute-class.yaml b/helm-overrides/gke-datascience-prd-as1a/computeclass/g2-standard-16-l4-compute-class.yaml deleted file mode 100644 index b0e7467..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/computeclass/g2-standard-16-l4-compute-class.yaml +++ /dev/null @@ -1,49 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-16-l4-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - serviceAccount: gke-gpu-node-sa@meesho-dsci-as1-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-south1-a - machineType: g2-standard-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-south1-b - machineType: g2-standard-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-south1-c - machineType: g2-standard-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-as1a/computeclass/g2-standard-4-l4-compute-class.yaml b/helm-overrides/gke-datascience-prd-as1a/computeclass/g2-standard-4-l4-compute-class.yaml deleted file mode 100644 index 046307f..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/computeclass/g2-standard-4-l4-compute-class.yaml +++ /dev/null @@ -1,49 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-4-l4-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - serviceAccount: gke-gpu-node-sa@meesho-dsci-as1-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-south1-a - machineType: g2-standard-4 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-south1-b - machineType: g2-standard-4 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-south1-c - machineType: g2-standard-4 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-as1a/computeclass/g2-standard-8-l4-300gb-compute-class.yaml b/helm-overrides/gke-datascience-prd-as1a/computeclass/g2-standard-8-l4-300gb-compute-class.yaml deleted file mode 100644 index 29f7685..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/computeclass/g2-standard-8-l4-300gb-compute-class.yaml +++ /dev/null @@ -1,52 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-8-l4-300gb-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - serviceAccount: gke-gpu-node-sa@meesho-dsci-as1-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-south1-a - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskSize: 300 - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-south1-b - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskSize: 300 - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-south1-c - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskSize: 300 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-as1a/computeclass/g2-standard-8-l4-compute-class.yaml b/helm-overrides/gke-datascience-prd-as1a/computeclass/g2-standard-8-l4-compute-class.yaml deleted file mode 100644 index abbec8f..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/computeclass/g2-standard-8-l4-compute-class.yaml +++ /dev/null @@ -1,49 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-8-l4-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - serviceAccount: gke-gpu-node-sa@meesho-dsci-as1-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-south1-a - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-south1-b - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-south1-c - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-as1a/computeclass/n2d-standard-48-4lssd-compute-class.yaml b/helm-overrides/gke-datascience-prd-as1a/computeclass/n2d-standard-48-4lssd-compute-class.yaml deleted file mode 100644 index 0b58afd..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/computeclass/n2d-standard-48-4lssd-compute-class.yaml +++ /dev/null @@ -1,25 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: n2d-standard-48-4lssd-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - labels: - dedicated: n2d-standard-48-4lssd - serviceAccount: sa-dsci-shared-nodepool-prd@meesho-dsci-as1-prd-0622.iam.gserviceaccount.com - priorities: - - localStorage: - ssdCount: 4 - location: - zones: - - asia-south1-a - machineType: n2d-standard-48 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-as1a/conntrack-adjuster/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 4d0df09..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,20 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - max: 2097152 - hashsize: 524288 - sleepInterval: 30 - -nodeAffinity: - enabled: true - nodeSelectorTerms: - - matchExpressions: - - key: cloud.google.com/compute-class - operator: In - values: - - "contour-internal-0" - - "contour-internal-1-c4d" - - "contour-internal-0-c4d" - - "contour-intra-1" - - "megaquad" diff --git a/helm-overrides/gke-datascience-prd-as1a/contour-ca-issuer/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/contour-ca-issuer/custom-values.yaml deleted file mode 100644 index 5e5d014..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/contour-ca-issuer/custom-values.yaml +++ /dev/null @@ -1,17 +0,0 @@ -# ClusterIssuer override for k8s-datascience-prd-ase1 (prd datascience cluster, ase1a zone). -# ClusterIssuer is a cluster-scoped resource — this file pins the issuer -# identity for this cluster so the name is auditable per-cluster. -# -# Must match the `issuerRef.name` in consuming Certificate CRs (see -# devops-helm-charts/2.0.0/templates/proxyless-grpc-cert.yaml). - -issuerName: contour-datascience-prd-ca-issuer -rootCASecretName: contour-datascience-ca - -externalSecret: - enabled: true - vaultPath: meesho/devops/contour/root-ca - refreshInterval: "0" - secretStoreRef: - name: vault-backend -namespace: cert-manager-datascience-prd \ No newline at end of file diff --git a/helm-overrides/gke-datascience-prd-as1a/contour-cert-checker/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/contour-cert-checker/custom-values.yaml deleted file mode 100644 index c71fb7f..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=k8s-datascience-prd-ase1"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - cloud.google.com/compute-class: datascience-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/gke-datascience-prd-as1a/contour-internal-0/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/contour-internal-0/custom-values.yaml deleted file mode 100644 index 30d55e4..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,118 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 60s - max-connection-duration: 600s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 500m - memory: 3Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - tcpLB: true - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 5% - maxUnavailable: 0 - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-0-arm - nodeSelector: - cloud.google.com/compute-class: contour-internal-0-arm - logLevel: error - extraArgs: - - '--concurrency 6' - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 10Gi - limits: - cpu: 6 - memory: 12Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-dsci-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/gke-datascience-prd-as1a/contour-internal-1/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/contour-internal-1/custom-values.yaml deleted file mode 100644 index b4d3d26..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,118 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 400s - max-connection-duration: 540s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 8 - memory: 10Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - tcpLB: true - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-1-arm - nodeSelector: - cloud.google.com/compute-class: contour-internal-1-arm - logLevel: error - extraArgs: - - '--concurrency 14' - terminationGracePeriodSeconds: 500 - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 500 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 600 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 50 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "25" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 14 - memory: 8Gi - limits: - cpu: 14 - memory: 27Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int1-dsci-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/gke-datascience-prd-as1a/contour-internal-dataproc/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/contour-internal-dataproc/custom-values.yaml deleted file mode 100644 index fbbf78a..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/contour-internal-dataproc/custom-values.yaml +++ /dev/null @@ -1,112 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 60s - max-connection-duration: 600s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 500m - memory: 3Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-dataproc" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 5% - maxUnavailable: 0 - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-dataproc-arm - nodeSelector: - cloud.google.com/compute-class: contour-internal-dataproc-arm - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 14' - resources: - requests: - cpu: 14 - memory: 8Gi - limits: - cpu: 14 - memory: 27Gi - service: - tcpLB: true - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/gke-datascience-prd-as1a/contour-internal-intra-0/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index 2d11141..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,114 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 60s - max-connection-duration: 600s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 500m - memory: 3Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 5% - maxUnavailable: 0 - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-intra-0-arm - nodeSelector: - cloud.google.com/compute-class: contour-internal-intra-0-arm - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 14' - resources: - requests: - cpu: 14 - memory: 8Gi - limits: - cpu: 14 - memory: 27Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/gke-datascience-prd-as1a/contour-internal-intra-1/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index 89871cd..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,115 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 400s - max-connection-duration: 540s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 8 - memory: 10Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-intra-1-arm - nodeSelector: - cloud.google.com/compute-class: contour-internal-intra-1-arm - logLevel: error - terminationGracePeriodSeconds: 500 - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 20' - resources: - requests: - cpu: 20 - memory: 12Gi - limits: - cpu: 30 - memory: 57Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/gke-datascience-prd-as1a/coredns/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/coredns/custom-values.yaml deleted file mode 100644 index 5958751..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/coredns/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -replicaCount: 40 -communicationType: "intra" - -labels: - bu: datascience - team: datascience-devops - env: prd - -clusterIP: 10.137.16.2 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - -nodeSelector: - cloud.google.com/compute-class: datascience-devops - kubernetes.io/os: linux diff --git a/helm-overrides/gke-datascience-prd-as1a/coroot-node-agent/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/coroot-node-agent/custom-values.yaml deleted file mode 100644 index 932f4dc..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/coroot-node-agent/custom-values.yaml +++ /dev/null @@ -1,52 +0,0 @@ -fullnameOverride: "coroot-node-agent-datascience-prd" -priorityClassName: "low-priority" - -scrape: "false" # disable scraping for now - -resources: - requests: - cpu: "300m" - memory: "1Gi" - limits: - cpu: "500m" - memory: "1.5Gi" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-datascience-contour-ext - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-datascience-default-prd-ase1 - - key: cloud.google.com/compute-class - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external \ No newline at end of file diff --git a/helm-overrides/gke-datascience-prd-as1a/external-secrets/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/external-secrets/custom-values.yaml deleted file mode 100644 index 29867ec..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/external-secrets/custom-values.yaml +++ /dev/null @@ -1,61 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Annotations to add to Pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - nodeSelector: - cloud.google.com/compute-class: datascience-devops - certController: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - nodeSelector: - cloud.google.com/compute-class: datascience-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - webhook: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - nodeSelector: - cloud.google.com/compute-class: datascience-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" \ No newline at end of file diff --git a/helm-overrides/gke-datascience-prd-as1a/flagger/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/flagger/custom-values.yaml deleted file mode 100644 index 576e726..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/flagger/custom-values.yaml +++ /dev/null @@ -1,62 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - cloud.google.com/compute-class: datascience-devops - -tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: datascience-devops - bu: infra - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-datascience-rollout-service.prd-datascience-rollout-service.svc.cluster.local" - -rolloutService: - enabled: true - apps: diff --git a/helm-overrides/gke-datascience-prd-as1a/fluentd/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/fluentd/custom-values.yaml deleted file mode 100644 index 3c92cfe..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/fluentd/custom-values.yaml +++ /dev/null @@ -1,731 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-dssre-fluentd-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: datascience - team: sre - type: fluentd - service: fluentd-datascience-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: datascience - team: sre - type: fluentd - service: fluentd-datascience-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: - - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - from_encoding "utf-8" - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/gke-datascience-prd-as1a/ingress-nginx-internal/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/ingress-nginx-internal/custom-values.yaml deleted file mode 100644 index 7b5a032..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/ingress-nginx-internal/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 6 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dsci-int-prd"}}}' - nodeSelector: - cloud.google.com/compute-class: nginx-internal - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "nginx-internal" - effect: "NoSchedule" diff --git a/helm-overrides/gke-datascience-prd-as1a/keda/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/keda/custom-values.yaml deleted file mode 100644 index e710795..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/keda/custom-values.yaml +++ /dev/null @@ -1,46 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - cloud.google.com/compute-class: datascience-devops - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - podLabels: - bu: "datascience" - team: "datascience-devops" - metricsAdapter: - bu: "datascience" - team: "datascience-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 500m - memory: 450Mi - # -- Manage [resource request & limits] of KEDA metrics apiserver pod - metricServer: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 300m - memory: 200Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 100m - memory: 180Mi - requests: - cpu: 60m - memory: 120Mi - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' diff --git a/helm-overrides/gke-datascience-prd-as1a/kube-dns/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/kube-dns/custom-values.yaml deleted file mode 100644 index 4563673..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.137.16.2"],"prd.mrouter.int.svc.cluster.local":["10.137.16.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.137.16.2"]} diff --git a/helm-overrides/gke-datascience-prd-as1a/kube-events/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/kube-events/custom-values.yaml deleted file mode 100644 index bcaf8d1..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-datascience-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - cloud.google.com/compute-class: datascience-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: datascience-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: datascience-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: datascience-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: datascience-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: datascience-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/gke-datascience-prd-as1a/kube-state-metrics/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 28bac57..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-south1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-datascience-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-datascience-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "datascience" - team: "datascience-sre" - service: "kube-state-metrics-datascience-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - cloud.google.com/compute-class: "vmselect" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 17m - memory: 1.5Gi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/gke-datascience-prd-as1a/kubectl-mcp-server/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/kubectl-mcp-server/custom-values.yaml deleted file mode 100644 index 7d80dca..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/kubectl-mcp-server/custom-values.yaml +++ /dev/null @@ -1,117 +0,0 @@ -fullnameOverride: "kubectl-mcp-server" - -replicas: 1 - -image: - # In-house hardened build (helm-templates/kubectl-mcp-server/docker/) — - # NOT the upstream Docker Hub image. 266→18 HIGH/CRIT vulns, non-root, - # multi-stage slim base, kubectl v1.33.12 / helm v3.21.0. - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/prd/devop/kubectl-mcp-server - tag: "v2" - pullPolicy: IfNotPresent - -labels: - bu: infra - team: devops - service: kubectl-mcp-server - env: prd - -serviceAccount: - create: true - annotations: {} - -rbac: - create: true - -podAnnotations: {} - -podSecurityContext: {} - -securityContext: - runAsNonRoot: true - runAsUser: 1000 - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -priorityClassName: "" - -nodeSelector: - cloud.google.com/compute-class: "datascience-devops" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: NoSchedule - -affinity: {} - -resources: - requests: - cpu: 250m - memory: 512Mi - limits: - cpu: 1000m - memory: 1Gi - -mcp: - mode: single - transport: http - host: "0.0.0.0" - port: 8000 - -auth: - allowAnonymous: false - -externalSecrets: - enabled: true - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/prd/cntr/devop/kubectl-mcp-server-datascience" - -# tcpSocket on purpose — streamable-http transport exposes only /mcp, -# there is no /health route (see chart values.yaml note). -livenessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - -readinessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - -service: - port: 8000 - type: ClusterIP - -# Contour standard (2.0.0-style): renders HTTPProxy parent/child + -intra -# instead of a networking.k8s.io Ingress. -createContourGateway: true - -ingress: - enabled: true - ingressClassName: contour-internal-1 - servicePortNumber: 8000 - hosts: - - host: kubectl-mcp-server-datascience.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 diff --git a/helm-overrides/gke-datascience-prd-as1a/kubernetes-dashboard/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/kubernetes-dashboard/custom-values.yaml deleted file mode 100644 index cd5b892..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/kubernetes-dashboard/custom-values.yaml +++ /dev/null @@ -1,442 +0,0 @@ -# Copyright 2017 The Kubernetes Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -# General configuration shared across resources -app: - # Mode determines if chart should deploy a full Dashboard with all containers or just the API. - # - dashboard - deploys all the containers - # - api - deploys just the API - mode: 'dashboard' - image: - pullPolicy: IfNotPresent - pullSecrets: [] - scheduling: - # Node labels for pod assignment - # Ref: https://kubernetes.io/docs/user-guide/node-selection/ - nodeSelector: {} - security: - # Allow overriding csrfKey used by API/Auth containers. - # It has to be base64 encoded random 256 bytes string. - # If empty, it will be autogenerated. - csrfKey: ~ - # SecurityContext to be added to pods - # To disable set the following configuration to null: - # securityContext: null - securityContext: - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - # ContainerSecurityContext to be added to containers - # To disable set the following configuration to null: - # containerSecurityContext: null - containerSecurityContext: - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - runAsUser: 1001 - runAsGroup: 2001 - capabilities: - drop: ["ALL"] - # Pod Disruption Budget configuration - # Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - minAvailable: 0 - maxUnavailable: 0 - networkPolicy: - enabled: false - ingressDenyAll: false - # Raw network policy spec that overrides predefined spec - # Example: - # spec: - # egress: - # - ports: - # - port: 123 - spec: {} - - # Common labels & annotations shared across all deployed resources - labels: {} - annotations: {} - # Common priority class used for all deployed resources - priorityClassName: null - settings: - ## Global dashboard settings - global: - # # Cluster name that appears in the browser window title if it is set - # clusterName: "" - # # Max number of items that can be displayed on each list page - # itemsPerPage: 10 - # # Max number of labels that are displayed by default on most views. - # labelsLimit: 3 - # # Number of seconds between every auto-refresh of logs - # logsAutoRefreshTimeInterval: 5 - # # Number of seconds between every auto-refresh of every resource. Set 0 to disable - # resourceAutoRefreshTimeInterval: 10 - # # Hide all access denied warnings in the notification panel - # disableAccessDeniedNotifications: false - # # Hide all namespaces option in namespace selection dropdown to avoid accidental selection in large clusters thus preventing OOM errors - # hideAllNamespaces: false - # # Namespace that should be selected by default after logging in. - # defaultNamespace: default - # # List of namespaces that should be presented to user without namespace list privileges. - # namespaceFallbackList: - # - default - ## Pinned resources that will be displayed in dashboard's menu - pinnedResources: [] - # - kind: customresourcedefinition - # # Fully qualified name of a CRD - # name: prometheus.monitoring.coreos.com - # # Display name - # displayName: Prometheus - # # Is this CRD namespaced? - # namespaced: true - ingress: - enabled: false - hosts: - # Keep 'localhost' host only if you want to access Dashboard using 'kubectl port-forward ...' on: - # https://localhost:8443 - - localhost - # - kubernetes.dashboard.domain.com - ingressClassName: internal-nginx - # Use only if your ingress controllers support default ingress classes. - # If set to true ingressClassName will be ignored and not added to the Ingress resources. - # It should fall back to using IngressClass marked as the default. - useDefaultIngressClass: false - # This will append our Ingress with annotations required by our default configuration. - # nginx.ingress.kubernetes.io/backend-protocol: "HTTPS" - # nginx.ingress.kubernetes.io/ssl-passthrough: "true" - # nginx.ingress.kubernetes.io/ssl-redirect: "true" - useDefaultAnnotations: true - pathType: ImplementationSpecific - # If path is not the default (/), rewrite-target annotation will be added to the Ingress. - # It allows serving Kubernetes Dashboard on a sub-path. Make sure that the configured path - # does not conflict with gateway route configuration. - path: / - issuer: - name: selfsigned - # Scope determines what kind of issuer annotation will be used on ingress resource - # - default - adds 'cert-manager.io/issuer' - # - cluster - adds 'cert-manager.io/cluster-issuer' - # - disabled - disables cert-manager annotations - scope: default - tls: - enabled: true - # If provided it will override autogenerated secret name - secretName: "" - labels: {} - annotations: {} - # Use the following toleration if Dashboard can be deployed on a tainted control-plane nodes - # - key: node-role.kubernetes.io/control-plane - # effect: NoSchedule - tolerations: [] - affinity: {} - -auth: - role: auth - image: - repository: docker.io/kubernetesui/dashboard-auth - tag: 1.2.2 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: auth - containerPort: 8000 - protocol: TCP - args: [] - env: [] - volumeMounts: - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for Auth related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -# API deployment configuration -api: - role: api - image: - repository: docker.io/kubernetesui/dashboard-api - tag: 1.10.1 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: api - containerPort: 8000 - protocol: TCP - # Additional container arguments - # Full list of arguments: https://github.com/kubernetes/dashboard/blob/master/docs/common/arguments.md - # args: - # - --system-banner="Welcome to the Kubernetes Dashboard" - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store exec logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for API related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -# WEB UI deployment configuration -web: - role: web - image: - repository: docker.io/kubernetesui/dashboard-web - tag: 1.6.0 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: web - containerPort: 8000 - protocol: TCP - # Additional container arguments - # Full list of arguments: https://github.com/kubernetes/dashboard/blob/master/docs/common/arguments.md - # args: - # - --system-banner="Welcome to the Kubernetes Dashboard" - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for WEB UI related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -### Metrics Scraper -### Container to scrape, store, and retrieve a window of time from the Metrics Server. -### refs: https://github.com/kubernetes/dashboard/tree/master/modules/metrics-scraper -metricsScraper: - enabled: true - role: metrics-scraper - image: - repository: docker.io/kubernetesui/dashboard-metrics-scraper - tag: 1.2.1 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - containerPort: 8000 - protocol: TCP - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - livenessProbe: - httpGet: - scheme: HTTP - path: / - port: 8000 - initialDelaySeconds: 30 - timeoutSeconds: 30 - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for Metrics Scraper related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -## Optional Metrics Server sub-chart configuration -## Enable this if you don't already have metrics-server enabled on your cluster and -## want to use it with dashboard metrics-scraper -## refs: -## - https://github.com/kubernetes-sigs/metrics-server -## - https://github.com/kubernetes-sigs/metrics-server/tree/master/charts/metrics-server -metrics-server: - enabled: false - args: - - --kubelet-preferred-address-types=InternalIP - - --kubelet-insecure-tls - -## Required Kong sub-chart with DBless configuration to act as a gateway -## for our all containers. -kong: - enabled: false - admin: - tls: - enabled: false - ## Configuration reference: https://docs.konghq.com/gateway/3.6.x/reference/configuration - env: - dns_order: LAST,A,CNAME,AAAA,SRV - plugins: 'off' - nginx_worker_processes: 1 - ingressController: - enabled: false - manager: - enabled: false - dblessConfig: - configMap: kong-dbless-config - proxy: - type: ClusterIP - http: - enabled: true - -## Optional Cert Manager sub-chart configuration -## Enable this if you don't already have cert-manager enabled on your cluster. -cert-manager: - enabled: false - installCRDs: true - -## Optional Nginx Ingress sub-chart configuration -## Enable this if you don't already have nginx-ingress enabled on your cluster. -nginx: - enabled: false - controller: - electionID: ingress-controller-leader - ingressClassResource: - name: internal-nginx - default: false - controllerValue: k8s.io/internal-ingress-nginx - service: - type: ClusterIP - -## Extra configurations: -## - manifests -## - predefined roles -## - prometheus -## - etc... -extras: - # Extra Kubernetes manifests to be deployed - # manifests: - # - apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: additional-configmap - # data: - # mykey: myvalue - manifests: [] - serviceMonitor: - # Whether to create a Prometheus Operator service monitor. - enabled: false - # Here labels can be added to the serviceMonitor - labels: {} - # Here annotations can be added to the serviceMonitor - annotations: {} - # metrics.serviceMonitor.metricRelabelings Specify Metric Relabelings to add to the scrape endpoint - # ref: https://github.com/coreos/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # metrics.serviceMonitor.relabelings [array] Prometheus relabeling rules - relabelings: [] - # ServiceMonitor connection scheme. Defaults to HTTPS. - scheme: https - # ServiceMonitor connection tlsConfig. Defaults to {insecureSkipVerify:true}. - tlsConfig: - insecureSkipVerify: true diff --git a/helm-overrides/gke-datascience-prd-as1a/kyverno/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/kyverno/custom-values.yaml deleted file mode 100644 index e7593a2..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/kyverno/custom-values.yaml +++ /dev/null @@ -1,2240 +0,0 @@ -global: - - # -- Internal settings used with `helm template` to generate install manifest - # @ignored - templating: - enabled: false - debug: false - version: ~ - - image: - # -- (string) Global value that allows to set a single image registry across all deployments. - # When set, it will override any values set under `.image.registry` across the chart. - registry: ~ - # -- (list) Global list of Image pull secrets - # When set, it will override any values set under `imagePullSecrets` under different components across the chart. - imagePullSecrets: [] - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - caCertificates: - # -- Global CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - # Individual controller values will override this global value - data: ~ - - # -- Global value to set single volume to be mounted for CA certificates for all deployments. - # Not used when `.Values.global.caCertificates.data` is defined - # Individual controller values will override this global value - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Additional container environment variables to apply to all containers and init containers - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Global node labels for pod assignment. Non-global values will override the global value. - nodeSelector: - dedicated: datascience-kyverno - - # -- Global List of node taints to tolerate. Non-global values will override the global value. - tolerations: - - key: "dedicated" - operator: "Equal" - value: "datascience-kyverno" - effect: "NoSchedule" - -# -- (string) Override the name of the chart -nameOverride: ~ - -# -- (string) Override the expanded name of the chart -fullnameOverride: ~ - -# -- (string) Override the namespace the chart deploys to -namespaceOverride: ~ - -upgrade: - # -- Upgrading from v2 to v3 is not allowed by default, set this to true once changes have been reviewed. - fromV2: false - -apiVersionOverride: - # -- (string) Override api version used to create `PodDisruptionBudget`` resources. - # When not specified the chart will check if `policy/v1/PodDisruptionBudget` is available to - # determine the api version automatically. - podDisruptionBudget: ~ - -rbac: - roles: - # -- Aggregate ClusterRoles to Kubernetes default user-facing roles. For more information, see [User-facing roles](https://kubernetes.io/docs/reference/access-authn-authz/rbac/#user-facing-roles) - aggregate: - admin: true - view: true - -# Use openreports.io as the API group for reporting -openreports: - # -- Enable OpenReports feature in controllers - enabled: false - # -- Whether to install CRDs from the upstream OpenReports chart. Setting this to true requires enabled to also be true. - installCrds: false - -# CRDs configuration -crds: - - # -- Whether to have Helm install the Kyverno CRDs, if the CRDs are not installed by Helm, they must be added before policies can be created - install: true - - reportsServer: - # -- Kyverno reports-server is used in your cluster - enabled: false - - groups: - - # -- Install CRDs in group `kyverno.io` - kyverno: - cleanuppolicies: true - clustercleanuppolicies: true - clusterpolicies: true - globalcontextentries: true - policies: true - policyexceptions: true - updaterequests: true - - # -- Install CRDs in group `policies.kyverno.io` - policies: - validatingpolicies: true - policyexceptions: true - imagevalidatingpolicies: true - namespacedimagevalidatingpolicies: true - mutatingpolicies: true - generatingpolicies: true - deletingpolicies: true - namespaceddeletingpolicies: true - namespacedvalidatingpolicies: true - - # -- Install CRDs in group `reports.kyverno.io` - reports: - clusterephemeralreports: true - ephemeralreports: true - - # -- Install CRDs in group `wgpolicyk8s.io` - wgpolicyk8s: - clusterpolicyreports: false ##lk - policyreports: false ##lk - - # -- Additional CRDs annotations (Replace=true avoids 262144-byte annotation limit on large CRDs) - annotations: - argocd.argoproj.io/sync-options: Replace=true - # strategy.spinnaker.io/replace: 'true' - - # -- Additional CRDs labels - customLabels: {} - - migration: - - # -- Enable CRDs migration using helm post upgrade hook - enabled: true - - # -- Resources to migrate - resources: - - cleanuppolicies.kyverno.io - - clustercleanuppolicies.kyverno.io - - clusterpolicies.kyverno.io - - globalcontextentries.kyverno.io - - policies.kyverno.io - - policyexceptions.kyverno.io - - updaterequests.kyverno.io - - deletingpolicies.policies.kyverno.io - - generatingpolicies.policies.kyverno.io - - imagevalidatingpolicies.policies.kyverno.io - - namespacedimagevalidatingpolicies.policies.kyverno.io - - mutatingpolicies.policies.kyverno.io - - namespaceddeletingpolicies.policies.kyverno.io - - namespacedvalidatingpolicies.policies.kyverno.io - - policyexceptions.policies.kyverno.io - - validatingpolicies.policies.kyverno.io - - image: - # -- (string) Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- (string) Image repository - repository: kyverno/kyverno-cli - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- (string) Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podResources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -# Configuration -config: - - # -- Create the configmap. - create: true - - # -- Preserve the configmap settings during upgrade. - preserve: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - # -- Enable registry mutation for container images. Enabled by default. - enableDefaultRegistryMutation: true - - # -- The registry hostname used for the image mutation. - defaultRegistry: docker.io - - # -- Exclude groups - excludeGroups: - - system:nodes - - # -- Exclude usernames - excludeUsernames: [] - # - '!system:kube-scheduler' - - # -- Exclude roles - excludeRoles: [] - - # -- Exclude roles - excludeClusterRoles: [] - - # -- Generate success events. - generateSuccessEvents: false - - # -- Resource types to be skipped by the Kyverno policy engine. - # Make sure to surround each entry in quotes so that it doesn't get parsed as a nested YAML list. - # These are joined together without spaces, run through `tpl`, and the result is set in the config map. - # @default -- See [values.yaml](values.yaml) - resourceFilters: - - '[Event,*,*]' - # - '[*/*,kube-system,*]' - - '[*/*,kube-public,*]' - - '[*/*,kube-node-lease,*]' - - '[Node,*,*]' - - '[Node/?*,*,*]' - - '[APIService,*,*]' - - '[APIService/?*,*,*]' - - '[TokenReview,*,*]' - - '[SubjectAccessReview,*,*]' - - '[SelfSubjectAccessReview,*,*]' - - '[Binding,*,*]' - - '[Pod/binding,*,*]' - - '[ReplicaSet,*,*]' - - '[ReplicaSet/?*,*,*]' - - '[EphemeralReport,*,*]' - - '[ClusterEphemeralReport,*,*]' - # exclude resources from the chart - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:additional]' - - '[ClusterRoleBinding,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.configMapName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.metricsConfigMapName" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Job,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[Job/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.admission-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.background-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - # -- Sets the threshold for the total number of UpdateRequests generated for mutateExisitng and generate policies. - updateRequestThreshold: 1000 - - # -- Defines the `namespaceSelector`/`objectSelector` in the webhook configurations. - # The Kyverno namespace is excluded if `excludeKyvernoNamespace` is `true` (default) - webhooks: - # Exclude namespaces - # namespaceSelector: lk-s2 - # matchExpressions: - # - key: kubernetes.io/metadata.name - # operator: NotIn - # values: - # - kube-systemm lk-s2 - # Exclude objects - # objectSelector: - # matchExpressions: - # - key: webhooks.kyverno.io/exclude - # operator: DoesNotExist - - # -- Defines annotations to set on webhook configurations. - webhookAnnotations: - # Example to disable admission enforcer on AKS: - # 'admissions.enforcer/disabled': 'true' lk-s3 - - # -- Defines labels to set on webhook configurations. - webhookLabels: {} - # Example to adopt webhook resources in ArgoCD: - # 'argocd.argoproj.io/instance': 'kyverno' - - # -- Defines match conditions to set on webhook configurations (requires Kubernetes 1.27+). - matchConditions: [] - - # -- Exclude Kyverno namespace - # Determines if default Kyverno namespace exclusion is enabled for webhooks and resourceFilters - excludeKyvernoNamespace: true - - # -- resourceFilter namespace exclude - # Namespaces to exclude from the default resourceFilters - resourceFiltersExcludeNamespaces: [] - - # -- resourceFilters exclude list - # Items to exclude from config.resourceFilters - resourceFiltersExclude: [] - - # -- resourceFilter namespace include - # Namespaces to include to the default resourceFilters - resourceFiltersIncludeNamespaces: [] - - # -- resourceFilters include list - # Items to include to config.resourceFilters - resourceFiltersInclude: [] - -# Metrics configuration -metricsConfig: - - # -- Create the configmap. - create: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - namespaces: - - # -- List of namespaces to capture metrics for. - include: [] - - # -- list of namespaces to NOT capture metrics for. - exclude: [] - - # -- (string) Rate at which metrics should reset so as to clean up the memory footprint of kyverno metrics, if you might be expecting high memory footprint of Kyverno's metrics. Default: 0, no refresh of metrics. WARNING: This flag is not working since Kyverno 1.8.0 - metricsRefreshInterval: ~ - # metricsRefreshInterval: 24h - - # -- (list) Configures the bucket boundaries for all Histogram metrics, changing this configuration requires restart of the kyverno admission controller - bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 15, 20, 25, 30] - - # -- (map) Configures the exposure of individual metrics, by default all metrics and all labels are exported, changing this configuration requires restart of the kyverno admission controller - metricsExposure: - kyverno_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_image_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_mutating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_generating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_admission_review_duration_seconds: - # enabled: false - disabledLabelDimensions: ["resource_namespace"] - kyverno_policy_rule_info_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_policy_results_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_admission_requests_total: - disabledLabelDimensions: ["resource_namespace"] - kyverno_cleanup_controller_deletedobjects_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - -# -- Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -imagePullSecrets: {} - # regcred: - # registry: foo.example.com - # username: foobar - # password: secret - # regcred2: - # registry: bar.example.com - # username: barbaz - # password: secret2 - -# -- Existing Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -existingImagePullSecrets: [] - # - test-registry - # - other-test-registry - -# Tests configuration -test: - # -- Sleep time before running test - sleep: 20 - - image: - # -- (string) Image registry - registry: curlimages - # -- Image repository - repository: curl - # -- Image tag - # Defaults to `latest` if omitted - tag: '8.10.1' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - # -- Security context for the test containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- Additional Pod annotations - podAnnotations: {} - - # -- List of node taints to tolerate - tolerations: [] - -# -- Additional labels -customLabels: {} - - -webhooksCleanup: - # -- Create a helm pre-delete hook to cleanup webhooks. - enabled: true - - autoDeleteWebhooks: - # -- Allow webhooks controller to delete webhooks using finalizers - enabled: false - - image: - # -- (string) Image registry - registry: registry.k8s.io - # -- Image repository - repository: kubectl - # -- Image tag - # Defaults to `latest` if omitted - tag: 'v1.32.7' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -grafana: - # -- Enable grafana dashboard creation. - enabled: false - - # -- Configmap name template. - configMapName: '{{ include "kyverno.fullname" . }}-grafana' - - # -- (string) Namespace to create the grafana dashboard configmap. - # If not set, it will be created in the same namespace where the chart is deployed. - namespace: ~ - - # -- Grafana dashboard configmap annotations. - annotations: {} - - # -- Grafana dashboard configmap labels - labels: - grafana_dashboard: "1" - - # -- create GrafanaDashboard custom resource referencing to the configMap. - # according to https://grafana-operator.github.io/grafana-operator/docs/examples/dashboard_from_configmap/readme/ - grafanaDashboard: - create: false - folder: kyverno - allowCrossNamespaceImport: true - matchLabels: - dashboards: "grafana" - -# Features configuration -features: - admissionReports: - # -- Enables the feature - enabled: true - aggregateReports: - # -- Enables the feature - enabled: true - policyReports: - # -- Enables the feature - enabled: false ##lk - validatingAdmissionPolicyReports: - # -- Enables the feature - enabled: true - mutatingAdmissionPolicyReports: - # -- Enables the feature - enabled: false - reporting: - # -- Enables the feature - validate: true - # -- Enables the feature - mutate: true - # -- Enables the feature - mutateExisting: true - # -- Enables the feature - imageVerify: true - # -- Enables the feature - generate: true - autoUpdateWebhooks: - # -- Enables the feature - enabled: true - backgroundScan: - # -- Enables the feature - enabled: true - # -- Number of background scan workers - backgroundScanWorkers: 2 - # -- Background scan interval - backgroundScanInterval: 1h - # -- Skips resource filters in background scan - skipResourceFilters: true - configMapCaching: - # -- Enables the feature - enabled: true - controllerRuntimeMetrics: - # -- Bind address for controller-runtime metrics (use "0" to disable it) - bindAddress: ":8080" - deferredLoading: - # -- Enables the feature - enabled: true - dumpPayload: - # -- Enables the feature - enabled: false - forceFailurePolicyIgnore: - # -- Enables the feature - enabled: false - generateValidatingAdmissionPolicy: - # -- Enables the feature - enabled: true - generateMutatingAdmissionPolicy: - # -- Enables the feature - enabled: false - dumpPatches: - # -- Enables the feature - enabled: false - globalContext: - # -- Maximum allowed response size from API Calls. A value of 0 bypasses checks (not recommended) - maxApiCallResponseLength: 2000000 - logging: - # -- Logging format - format: text - # -- Logging verbosity - verbosity: 2 - omitEvents: - # -- Events which should not be emitted (possible values `PolicyViolation`, `PolicyApplied`, `PolicyError`, and `PolicySkipped`) - eventTypes: - - PolicyApplied - - PolicySkipped - # - PolicyViolation - # - PolicyError - policyExceptions: - # -- Enables the feature - enabled: true - # -- Restrict policy exceptions to a single namespace - # Set to "*" to allow exceptions in all namespaces - namespace: '' - protectManagedResources: - # -- Enables the feature - enabled: false - registryClient: - # -- Allow insecure registry - allowInsecure: false - # -- Enable registry client helpers - credentialHelpers: - - default - - google - - amazon - - azure - - github - ttlController: - # -- Reconciliation interval for the label based cleanup manager - reconciliationInterval: 1m - tuf: - # -- Enables the feature - enabled: false - # -- (string) Path to Tuf root - root: ~ - # -- (string) Raw Tuf root - rootRaw: ~ - # -- (string) Tuf mirror - mirror: ~ - -# Admission controller configuration -admissionController: - autoscaling: - # -- Enable horizontal pod autoscaling - enabled: true - - # -- Minimum number of pods - minReplicas: 1 - - # -- Maximum number of pods - maxReplicas: 10 - - # -- Target CPU utilization percentage - targetCPUUtilizationPercentage: 80 - - # -- Configurable scaling behavior - behavior: {} - - # -- Overrides features defined at the root level - featuresOverride: - admissionReports: - # -- Max number of admission reports allowed in flight until the admission controller stops creating new ones - backPressureThreshold: 1000 - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- The ServiceAccount name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno admission controller activities. - # This will help ensure Kyverno stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- admissionController webhook server port - # in case you are using hostNetwork: true, you might want to change the port the webhookServer is listening to - webhookServer: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - admission-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.admissionController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - initContainer: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyvernopre - # -- (string) Image tag - # If missing, defaults to image.tag - tag: ~ - # -- (string) Image pull policy - # If missing, defaults to image.pullPolicy - pullPolicy: ~ - - resources: - # -- Pod resource limits - # limits: - # cpu: 100m - # memory: 256Mi - # -- Pod resource requests - requests: - cpu: 1000m - memory: 1Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - container: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyverno - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - resources: - # -- Pod resource limits - # limits: - # memory: 384Mi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Array of extra init containers - extraInitContainers: [] - # - name: init-container - # image: busybox - # command: ['sh', '-c', 'echo Hello'] - - # -- Array of extra containers to run alongside kyverno - extraContainers: [] - # - name: myapp-container - # image: busybox - # command: ['sh', '-c', 'echo Hello && sleep 3600'] - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Kyverno's metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Background controller configuration -backgroundController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable background controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: - - apiGroups: - - networking.k8s.io - resources: - - ingresses - - ingressclasses - - networkpolicies - verbs: - - create - - update - - patch - - delete - - apiGroups: - - rbac.authorization.k8s.io - resources: - - rolebindings - - roles - verbs: - - create - - update - - patch - - delete - - apiGroups: - - '' - resources: - - configmaps - - resourcequotas - - limitranges - verbs: - - create - - update - - patch - - delete - - apiGroups: - - resource.k8s.io - resources: - - resourceclaims - - resourceclaimtemplates - verbs: - - create - - delete - - update - - patch - - deletecollection - - apiGroups: - - autoscaling - resources: - - horizontalpodautoscalers - verbs: - - get - - list - - watch - - update - - patch - - apiGroups: - - apps - resources: - - deployments - - statefulsets - - replicasets - verbs: - - get - - list - - watch - - update - - patch - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - # - patch - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/background-controller - # -- Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - # limits: - # memory: 128Mi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - background-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.backgroundController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - # -- backgroundController server port - # in case you are using hostNetwork: true, you might want to change the port the backgroundController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Cleanup controller configuration -cleanupController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable cleanup controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - delete - # - list - # - watch - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/cleanup-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- cleanupController server port - # in case you are using hostNetwork: true, you might want to change the port the cleanupController is listening to - server: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - # limits: - # memory: 128Mi - # -- Pod resource requests - requests: - cpu: 1100m - memory: 1.2Gi - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - cleanup-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `service.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Reports controller configuration -reportsController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable reports controller. - enabled: false ##lk - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/reports-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno reports controller activities. - # This will help ensure Kyverno reports stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - memory: 128Mi - # -- Pod resource requests - requests: - cpu: 100m - memory: 64Mi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - reports-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.reportsController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- (string) Service node port. - # Only used if `type` is `NodePort`. - nodePort: ~ - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- (string) Traces receiver address - address: ~ - # -- (string) Traces receiver port - port: ~ - # -- (string) Traces receiver credentials - creds: ~ - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- (string) Otel collector endpoint - collector: ~ - # -- (string) Otel collector credentials - creds: ~ - - # -- reportsController server port - # in case you are using hostNetwork: true, you might want to change the port the reportsController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - - # -- Enable sanity check for reports CRDs - sanityChecks: true diff --git a/helm-overrides/gke-datascience-prd-as1a/kyverno/policies/protect-namespace.yaml b/helm-overrides/gke-datascience-prd-as1a/kyverno/policies/protect-namespace.yaml deleted file mode 100644 index 5db5729..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/kyverno/policies/protect-namespace.yaml +++ /dev/null @@ -1,40 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: protect-critical-namespaces -spec: - validationFailureAction: Enforce - background: false - rules: - - name: deny-deletion-of-critical-namespaces - match: - any: - - resources: - kinds: - - Namespace - names: - - keda-datascience-prd - - contour-internal-0-datascience-prd - - contour-internal-0-datascience-prd-intra - - contour-external-datascience-prd - - external-secrets-datascience-prd - - flagger-datascience-prd - - victoriametrics - - kube-system - - monitoring - - telegraf-operator - - loadtester - - opentelemetry - - fluentd - - kube-events - - observability - - contour-cert-checker-ns - - aurva-dataplane - validate: - message: "Deletion of critical namespaces is not allowed." - deny: - conditions: - any: - - key: "{{request.operation}}" - operator: Equals - value: DELETE diff --git a/helm-overrides/gke-datascience-prd-as1a/kyverno/policies/restrict-replicas.yaml b/helm-overrides/gke-datascience-prd-as1a/kyverno/policies/restrict-replicas.yaml deleted file mode 100644 index d456985..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/kyverno/policies/restrict-replicas.yaml +++ /dev/null @@ -1,29 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: restrict-minimum-replicas-selected-ns -spec: - validationFailureAction: Enforce - background: false - rules: - - name: min-replicas-deployments-statefulsets - match: - resources: - kinds: - - Deployment - - StatefulSet - - Deployment/scale - - StatefulSet/scale - names: - - coredns - namespaces: - - kube-system - operations: - - CREATE - - UPDATE - validate: - message: "Deployments and StatefulSets must have at least 5 replicas in this namespace." - anyPattern: - # Case 1: replicas explicitly set and >= 5 - - spec: - replicas: ">=5" diff --git a/helm-overrides/gke-datascience-prd-as1a/loadtester/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/loadtester/custom-values.yaml deleted file mode 100644 index 906f85b..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/loadtester/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -replicaCount: 1 - -Namespace: loadtester - -image: - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/dev/devops/flagger-loadtester - tag: 0.30.0 - pullPolicy: IfNotPresent - pullSecret: - -podLabels: - bu: datascience - team: datascience-shared - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - -podPriorityClassName: "" - -logLevel: info -cmd: - timeout: 1h - namespaceRegexp: "" - -nameOverride: "loadtester" -fullnameOverride: "" - -env: [] - -service: - type: ClusterIP - port: 80 - -resources: - requests: - cpu: 10m - memory: 64Mi - -volumes: [] -volumeMounts: [] - -nodeSelector: - cloud.google.com/compute-class: datascience-devops - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - -affinity: {} - -ingress: - slowStart: - enabled: false - enabled: true - ingressClassName: contour-internal-1 - servicePort: http - hosts: - - host: datascience-flagger-loadtester.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / -createContourGateway: true -contourResponseTimeout: false -rbac: - # rbac.create: `true` if rbac resources should be created - create: true - # rbac.scope: `cluster` to create cluster-scope rbac resources (ClusterRole/ClusterRoleBinding) - # otherwise, namespace-scope rbac resources will be created (Role/RoleBinding) - scope: - # rbac.rules: array of rules to apply to the role. example: - # rules: - # - apiGroups: [""] - # resources: ["pods"] - # verbs: ["list", "get"] - rules: [] - -# name of an existing service account to use - if not creating rbac resources -serviceAccountName: "" - -# App Mesh virtual node settings (to be used for AppMesh v1beta1) -meshName: "" -#backends: -# - app1.namespace -# - app2.namespace - -# App Mesh virtual node settings (to be used for AppMesh v1beta2) -appmesh: - enabled: false - backends: - - podinfo - - podinfo-canary - -#Istio virtual service and gatway settings. TLS secrets should be in namespace before enbaled it. ( secret format loadtester.fullname ) -istio: - enabled: false - host: flagger-loadtester.flagger - gateway: - enabled: false - tls: - enabled: false - httpsRedirect: false - -# when enabled, it will add a security context for the loadtester pod -securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 100 - runAsGroup: 101 - -podDisruptionBudget: - enabled: false - minAvailable: 1 diff --git a/helm-overrides/gke-datascience-prd-as1a/opentelemetry-coralogix/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/opentelemetry-coralogix/custom-values.yaml deleted file mode 100644 index 30fdbf8..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/opentelemetry-coralogix/custom-values.yaml +++ /dev/null @@ -1,281 +0,0 @@ -global: - domain: "coralogixsg.com" - defaultApplicationName: "default" - defaultSubsystemName: "nodes" - - # Old endpoint based configuration, - # please use domain instead. - traces: - endpoint: "" - metrics: - endpoint: "" - logs: - endpoint: "" - -# set distribution to openshift for openshift clusters -distribution: "" - -# Opentelemetry collector configuration -dedicatedValue: false -schedulerName: default-scheduler - -labels: - bu: "datascience" - team: "sre" - service: "opentelemetry-datascience-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: false - key: prd/admin/coralogix-keys - secretStoreRef: - name: vault-backend - -image: - # If you want to use the core image `otel/opentelemetry-collector`, you also need to change `command.name` value to `otelcol`. - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - # repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - tag: "0.87.0" - # When digest is set to a non-empty value, images will be pulled by digest (regardless of tag value). - digest: "" -mode: daemonset -rollout: - rollingUpdate: - # When 'mode: daemonset', maxSurge cannot be used when hostPort is set for any of the ports - # maxSurge: 25% - maxUnavailable: 1 - strategy: RollingUpdate -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" -hostNetwork: true -dnsPolicy: "ClusterFirstWithHostNet" -fullnameOverride: opentelemetry-datascience-prd - -presets: - logsCollection: - enabled: false - storeCheckpoints: false - maxRecombineLogSize: 1048576 - extraFilelogOperators: [] -# - type: recombine -# combine_field: body -# source_identifier: attributes["log.file.path"] -# is_first_entry: body matches "^(YOUR-LOGS-REGEX)" - kubernetesAttributes: - enabled: false - hostMetrics: - enabled: false - kubeletMetrics: - enabled: false - -extraEnvs: -- name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" -- name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName -config: - extensions: {} - # zpages: - # endpoint: localhost:55679 - # pprof: - # endpoint: localhost:1777 - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 4s - tls: - insecure: true - sending_queue: - enabled: true - queue_size: 6000 - num_consumers: 200 - resolver: - dns: - hostname: opentelemetry-admin-prd.opentelemetry.svc.clusterset.local - # coralogix: - # timeout: "30s" - # private_key: "${CORALOGIX_PRIVATE_KEY}" - # domain: "{{.Values.global.domain}}" - # traces: - # endpoint: "{{ .Values.global.traces.endpoint }}" - # metrics: - # endpoint: "{{ .Values.global.metrics.endpoint }}" - # logs: - # endpoint: "{{ .Values.global.logs.endpoint }}" - # application_name_attributes: - # - "k8s.namespace.name" - # - "service.namespace" - # subsystem_name_attributes: - # - "k8s.deployment.name" - # - "k8s.statefulset.name" - # - "k8s.daemonset.name" - # - "k8s.cronjob.name" - # - "k8s.job.name" - # - "k8s.container.name" - # - "k8s.node.name" - # - "service.name" - # application_name: "{{.Values.global.defaultApplicationName }}" - # subsystem_name: "{{.Values.global.defaultSubsystemName }}" - processors: - k8sattributes: - filter: - node_from_env_var: KUBE_NODE_NAME - extract: - metadata: - - "k8s.namespace.name" - - "k8s.deployment.name" - - "k8s.statefulset.name" - - "k8s.daemonset.name" - - "k8s.cronjob.name" - - "k8s.job.name" - - "k8s.pod.name" - - "k8s.node.name" - memory_limiter: null # Will get the k8s resource limits - resourcedetection/env: - detectors: ["system","env"] - timeout: 2s - override: false - # spanmetrics: - # metrics_exporter: coralogix - # dimensions: - # - name: "k8s.deployment.name" - # - name: "k8s.statefulset.name" - # - name: "k8s.daemonset.name" - # - name: "k8s.cronjob.name" - # - name: "k8s.job.name" - # - name: "k8s.container.name" - # - name: "k8s.node.name" - # - name: "k8s.namespace.name" - receivers: - otlp: - protocols: - grpc: - endpoint: ${MY_POD_IP}:4317 - http: - endpoint: ${MY_POD_IP}:4318 - zipkin: - endpoint: ${MY_POD_IP}:9411 - jaeger: - protocols: - grpc: - endpoint: ${MY_POD_IP}:14250 - thrift_http: - endpoint: ${MY_POD_IP}:14268 - thrift_compact: - endpoint: ${MY_POD_IP}:6831 - thrift_binary: - endpoint: ${MY_POD_IP}:6832 - prometheus: - config: - scrape_configs: - - job_name: opentelemetry-collector - scrape_interval: 30s - static_configs: - - targets: - - ${MY_POD_IP}:8888 - service: - extensions: - # - zpages - # - pprof - - health_check -# - memory_ballast - telemetry: - metrics: - address: ${env:MY_POD_IP}:8888 - pipelines: - traces: - exporters: - - loadbalancing - processors: - - memory_limiter - # - spanmetrics - - batch - receivers: - - otlp - metrics: null - logs: null -tolerations: - - operator: Exists - -resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 1 - memory: 2G - -ports: - jaeger-binary: - enabled: true - containerPort: 6832 - servicePort: 6832 - hostPort: 6832 - protocol: TCP - # In order to enable podMonitor, following part must be enabled in order to expose the required port: - # metrics: - # enabled: true - -# podMonitor: -# enabled: true - -# prometheusRule: -# enabled: true -# defaultRules: -# enabled: true - -# Annotations to be added to pod -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-datascience-contour-ext - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-datascience-default-prd-ase1 - - key: cloud.google.com/compute-class - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert \ No newline at end of file diff --git a/helm-overrides/gke-datascience-prd-as1a/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index da13cf8..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,134 +0,0 @@ -fullnameOverride: opentelemetry-datascience-prd - -mode: daemonset - -# priorityClassName: "system-node-critical" -# Removing since dataengg pods are going into pending state -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-datascience-contour-ext - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-datascience-default-prd-ase1 - - key: cloud.google.com/compute-class - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external - -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 100m - memory: 124Mi - limits: - cpu: 200m - memory: 256Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "datascience" - team: "sre" - service: "opentelemetry-datascience-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/gke-datascience-prd-as1a/paused-container/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/paused-container/custom-values.yaml deleted file mode 100644 index 2797988..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/paused-container/custom-values.yaml +++ /dev/null @@ -1,82 +0,0 @@ -# Default values for hello-world. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -priorityClass: - name: "low-priority" - -image: - repository: nginx - tag: "1.14.2" - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - -nameOverride: "" -fullnameOverride: "paused-container-datascience-prd" - -extraLabels: - team: "devops" - bu: "datascience" - env: "prd" - service: "paused-container-datascience-prd" - priority: "p3" - type: "tools" - -topologySpreadConstraints: [] - # - labelSelector: - # matchLabels: - # cloud.google.com/compute-class: vminsert - # maxSkew: 1 - # topologyKey: topology.kubernetes.io/hostname - # whenUnsatisfiable: DoNotSchedule - -affinity: - podAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - - labelSelector: - matchExpressions: - - key: type - operator: In - values: - - vmselect - - vminsert - - tools - topologyKey: topology.kubernetes.io/zone - podAntiAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - - labelSelector: - matchExpressions: - - key: type - operator: In - values: - - vmselect - - vminsert - - tools - topologyKey: kubernetes.io/hostname - namespaceSelector: {} - - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" - -service: - type: ClusterIP - port: 80 - -deployments: - - nodepool: vminsert - bu: datascience - cpuRequest: 1 - memoryRequest: 1Gi - replicaCount: 3 - - nodepool: vmselect - bu: datascience - cpuRequest: 1 - memoryRequest: 1Gi - replicaCount: 3 diff --git a/helm-overrides/gke-datascience-prd-as1a/prometheus-node-exporter/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index f0ab538..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,496 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-dataengg-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "datascience" - team: "datascience-sre" - service: "node-exporter-datascience-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/gke-datascience-prd-as1a/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index 83e9788..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-datascience-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-datascience-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "datascience" - team: "datascience-sre" - service: "stackdriver-exporter-datascience-prd" - env: "prd" - priority: "p0" - type: "exporter" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-datascience-prd-0622" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,logging.googleapis.com/user,run.googleapis.com/,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime,compute.googleapis.com/instance' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - cloud.google.com/compute-class: vmselect - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-dssre-stackdriver-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/gke-datascience-prd-as1a/telegraf-operator/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/telegraf-operator/custom-values.yaml deleted file mode 100644 index 77aaa6b..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,150 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "datascience" - team: "datascience-sre" - service: "telegraf-operator-datascience-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - cloud.google.com/compute-class: "vmselect" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false diff --git a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-agent-dr/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-agent-dr/custom-values.yaml deleted file mode 100644 index a0c6400..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-agent-dr/custom-values.yaml +++ /dev/null @@ -1,295 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-datascience-prd-dr -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-datascience-prd-dr-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent-dr - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - minReadySeconds: 30 - progressDeadlineSeconds: 60 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-dssre-vmagent-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - - https://vminsert-infra-prd-dr.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vmagent-datascience-prd" - env: "prd" - priority: "p0" - type: "vmagent-dr" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-datascience-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: contour-internal-1 - annotations: {} - # nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - # nginx.ingress.kubernetes.io/rewrite-target: / - # nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-datascience-prd-dr.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 10 - memory: 10Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - cloud.google.com/compute-class: "vmagent-dr" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-agent-fb/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-agent-fb/custom-values.yaml deleted file mode 100644 index 885b50d..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-agent-fb/custom-values.yaml +++ /dev/null @@ -1,272 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 1 - -fullnameOverride: vmagent-datascience-prd-fb -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-datascience-prd-fb-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - -# Remove topology spread constraints -topologySpreadConstraints: [] - -# Modify deployment strategy -deployment: - enabled: true - -# Remove HPA -horizontalPodAutoscaler: - enabled: false - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-infr-sre-vmagent-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -# Remove pod disruption budget -podDisruptionBudget: - enabled: false - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-datascience.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vmsinglenode-infra-prd.victoriametrics.svc.clusterset.local:8428/api/v1/write - - https://vmsinglenode-infra-prd.meeshogcp.in/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "datascience-fb" - team: "sre" - service: "vmagent-datascience-prd-fb" - env: "prd" - priority: "p0" - type: "vmagent-fb" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-datascience-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-datascience-prd-fb.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -# Modify resources -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - requests: - cpu: 2 - memory: 5Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - cloud.google.com/compute-class: "vmselect" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-agent/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index 3c07578..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,295 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-datascience-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-datascience-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - minReadySeconds: 30 - progressDeadlineSeconds: 60 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-sre-vmagnt-prd-mds@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-datascience.meeshogcp.in/insert/multitenant/prometheus/api/v1/write - - http://vminsert-datascience-prd.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vmagent-datascience-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-datascience-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: contour-internal-1 - annotations: {} - # nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - # nginx.ingress.kubernetes.io/rewrite-target: / - # nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-datascience-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 40 - memory: 80Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - cloud.google.com/compute-class: "vmagent-mds" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-mds" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert-dr/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert-dr/custom-values.yaml deleted file mode 100644 index 30ec234..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert-dr/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert-dr - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-datascience-prd-dr - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/datascience/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-datascience-prd-dr.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-datascience-prd-dr" - env: "prd" - priority: "p0" - type: "vmalert-dr" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmagent-dr" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert-secured-stateful/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert-secured-stateful/custom-values.yaml deleted file mode 100644 index 4fc3021..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert-secured-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-stateful-secured-datascience-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-datascience-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-datascience-prd.victoriametrics.svc.cluster.local:8480/insert/200/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-datascience-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/sensitive/datascience/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: false - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-datascience-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "datascience" - team: "sre" - service: "vmalert-stateful-secured-datascience-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmselect" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-secured-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert-secured/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert-secured/custom-values.yaml deleted file mode 100644 index 276006a..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert-secured/custom-values.yaml +++ /dev/null @@ -1,305 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert-secured - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-secured-datascience-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-datascience-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-datascience-prd.victoriametrics.svc.cluster.local:8480/insert/200/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-datascience-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config1/sensitive/datascience/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-secured-datascience-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-secured-datascience-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmselect" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert-stateful/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert-stateful/custom-values.yaml deleted file mode 100644 index 633f205..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-datascience-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-datascience-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-datascience-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-datascience-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/datascience/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-datascience-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "datascience" - team: "sre" - service: "vmalert-datascience-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmselect" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert/custom-values.yaml deleted file mode 100644 index 737904e..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-alert/custom-values.yaml +++ /dev/null @@ -1,305 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-datascience-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-datascience-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-datascience-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-datascience-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config1/datascience/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-datascience-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-datascience-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmselect" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-insert/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index a0fab9f..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,234 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-datascience-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "high-priority" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-datascience-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vminsert-datascience-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 7 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - cloud.google.com/compute-class: "vminsert-mds" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 10 - memory: 10Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: true - # -- Ingress annotations - annotations: - # nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - # nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-datascience-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: contour-internal-0 - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-select/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index db31335..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,293 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-datascience-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "high-priority" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-datascience-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 150 - clusternative.maxConcurrentRequests: 256 - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vmselect-datascience-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 15 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - cloud.google.com/compute-class: "vmselect" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 20 - memory: 31Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: contour-internal-0 - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-datascience.prd.meesho.int - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-storage/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index 5ecbaab..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,316 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-datascience-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmstorage-sale-24aug" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - cloud.google.com/compute-class: "vmstorage-sale-24aug" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 250Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vmstorage-datascience-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 24 - memory: 330Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.93.7-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-datascience-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/gke-datascience-prd-as1a/victoriametrics-agent/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index ece2162..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,485 +0,0 @@ -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "datascience-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -# -- Replica count -replicaCount: 2 - -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - -# -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables -deployment: - spec: - minReadySeconds: 1200 - progressDeadlineSeconds: 1800 - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - -image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - # -- Image tag, set to `Chart.AppVersion` by default - tag: v1.133.0 # rewrites Chart.AppVersion - # -- Variant of the image to use. - # e.g. enterprise, scratch - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - -# -- Image pull secrets -imagePullSecrets: [] - -# -- Add additional DNS entries to pods hosts file. Check [official documentation](https://kubernetes.io/docs/tasks/network/customize-hosts-file-for-pods/) -hostAliases: [] -# - ip: 192.168.1.1 -# hostNames: -# - test.example.com -# - another.example.net - -# -- Override chart name -nameOverride: "" - -# -- Override resources fullname -fullnameOverride: "vm-agent-datascience-prd" - -# -- Container working directory -containerWorkingDir: "/" - -rbac: - # -- Enables Role/RoleBinding creation - create: true - - # -- Role/RoleBinding annotations - annotations: {} - - # -- Role/RoleBinding labels - extraLabels: {} - - # -- If true and `rbac.enabled`, will deploy a Role/RoleBinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - - # -- additional rules for a role - extraRules: [] - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - # -- Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-sre-vmagnt-prd-mds@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # -- mount API token to pod directly - automountToken: true - -# -- See `kubectl explain poddisruptionbudget.spec` for more or check [official documentation](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# -- Generates `remoteWrite.*` flags and config maps with value content for values, that are of type list of map. -# Each item should contain `url` param to pass validation. -remoteWrite: - - url: http://vm-insert-datascience-prd.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write - disableOnDiskQueue: false - dropSamplesOnOverload: false - - url: http://prd-census-server-datascience.prd-census-server-datascience.svc.cluster.local/api/v1/write - disableOnDiskQueue: true - dropSamplesOnOverload: true -# urlRelabelConfig: -# - action: keep -# source_labels: [env] -# regex: "dev" - -# -- VMAgent extra command line arguments -extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8429 - promscrape.config.strictParse: false # default is true - promscrape.maxScrapeSize: 1000000000 # default is 167772160 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - # promscrape.suppressScrapeErrors: true # default is false - - # Uncomment and specify the port if you want to support any of the protocols: - # https://docs.victoriametrics.com/victoriametrics/vmagent/#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for more details. -env: - - name: GOGC - value: "200" - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth-secret - # key: password - -# -- Specify alternative source for env variables -envFrom: - [] - #- configMapRef: - # name: special-config - -# -- Extra labels for Deployment and Statefulset -extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-agent-datascience-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra labels for Pods only -podLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-agent-datascience-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra selector labels common for pod and service -selectorLabels: {} - -# -- Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# -- Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# -- Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -# -- Extra containers to run in a pod with vmagent -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -# -- Init containers for vmagent -initContainers: - [] - # - name: example - # image: example-image - -# -- Security context to be added to pod -podSecurityContext: - enabled: true - # fsGroup: 2000 - -# -- Security context to be added to pod's containers -securityContext: - enabled: true - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - # -- Enable agent service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - extraLabels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) for details - externalIPs: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8429 - # -- Target port - targetPort: http - # nodePort: 30000 - # -- Service type - type: ClusterIP - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - -ingress: - # -- Enable deployment of ingress for agent - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-agent-datascience-prd.meeshogcp.in - path: - - / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - - # -- Ingress controller class name - ingressClassName: "nginx-internal" - - # -- Ingress path type - pathType: Prefix - -# -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) -resources: - requests: - cpu: 70 - memory: 135Gi - -# -- Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) -nodeSelector: - cloud.google.com/compute-class: "vmagent-n4d" - -# -- Node tolerations for server scheduling to nodes with taints. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-n4d" - effect: "NoSchedule" - -# -- Pod topologySpreadConstraints -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# -- Pod affinity -affinity: {} - -# -- VMAgent [scraping configuration](https://docs.victoriametrics.com/victoriametrics/vmagent/#how-to-collect-metrics-in-prometheus-format) -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-datascience-prd" - -# -- Priority class to be assigned to the pod(s) -priorityClassName: "" - -# -- Enable the host network -hostNetwork: false - -serviceMonitor: - # -- Enable deployment of Service Monitor for server component. This is Prometheus operator object - enabled: false - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Service Monitor relabelings - relabelings: [] - # -- Basic auth params for Service Monitor - basicAuth: {} - # -- Service Monitor metricRelabelings - metricRelabelings: [] - # -- Service Monitor targetPort - targetPort: http - # interval: 15s - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - -# -- Empty dir configuration for a case, when persistence is disabled -emptyDir: {} - -persistentVolume: - # -- Create/use Persistent Volume Claim for server component. Empty dir if false - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Size of the volume. Should be calculated based on the logs you send and retention policy you set. - size: 10Gi - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume additional labels - extraLabels: {} - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Horizontal Pod Autoscaling. -# Note that it is not intended to be used for vmagents which perform scraping. -# In order to scale scraping vmagents check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) -horizontalPodAutoscaling: - # -- Use HPA for vmagent - enabled: false - # -- Maximum replicas for HPA to use to to scale vmagent - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale vmagent - minReplicas: 1 - # -- Metric for HPA to use to scale vmagent - metrics: [] - -# -- VMAgent scrape configuration -config: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -probe: - # -- Readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - # -- Liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - # -- Startup probe - startup: {} - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -allowedMetricsEndpoints: - - /metrics - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-datascience-prd-as1a/victoriametrics-insert/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/victoriametrics-insert/custom-values.yaml deleted file mode 100644 index ad3af27..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/victoriametrics-insert/custom-values.yaml +++ /dev/null @@ -1,411 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: false - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- IDs of vmstorage nodes to exclude from writing - excludeStorageIDs: [] - # -- Override default `app` label name - name: vminsert - # -- VMInsert strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - # override Chart.AppVersion - tag: v1.107.0-cluster - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMInsert http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-datascience-prd - # -- Extra command line arguments for vminsert component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8480 - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-datascience-prd-0.vm-storage-datascience-prd.victoriametrics.svc:8400" - - "vm-storage-datascience-prd-1.vm-storage-datascience-prd.victoriametrics.svc:8400" - - "vm-storage-datascience-prd-2.vm-storage-datascience-prd.victoriametrics.svc:8400" - - "vm-storage-datascience-prd-3.vm-storage-datascience-prd.victoriametrics.svc:8400" - - "vm-storage-datascience-prd-4.vm-storage-datascience-prd.victoriametrics.svc:8400" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-insert-datascience-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Pod's additional labels - podLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-insert-datascience-prd" - env: "prd" - priority: "p0" - type: "vminsert" - terminationGracePeriodSeconds: 30 - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - - # -- Readiness & Liveness probes - probe: - # -- VMInsert readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert startup probe - startup: {} - - # -- Relabel configuration - relabel: - enabled: false - config: [] - # -- Use existing configmap if specified - # otherwise .config values will be used. Relabel config **should** reside under `relabel.yml` key - configMap: "" - - # Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 7 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vminsert - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vminsert - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - cloud.google.com/compute-class: "vminsert-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 14 - memory: 10Gi - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - service: - # -- Create VMInsert service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here]( https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) for details - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-insert-datascience-prd.meeshogcp.in - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - - # -- Ingress controller class name - ingressClassName: nginx-internal - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for insert component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/insert" .Values.vminsert }}' - - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vminsert component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-datascience-prd-as1a/victoriametrics-select/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/victoriametrics-select/custom-values.yaml deleted file mode 100644 index 15d1b1a..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/victoriametrics-select/custom-values.yaml +++ /dev/null @@ -1,445 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - splitService: true - externalService: - enabled: true - name: vmselect-datascience-prd-proxy - # -- Override default `app` label name - clusternativeService: - enabled: false - name: "" - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # override Chart.AppVersion - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMSelect http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-datascience-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - # -- Extra command line arguments for vmselect component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8481 - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-datascience-prd-0.vm-storage-datascience-prd.victoriametrics.svc:8401" - - "vm-storage-datascience-prd-1.vm-storage-datascience-prd.victoriametrics.svc:8401" - - "vm-storage-datascience-prd-2.vm-storage-datascience-prd.victoriametrics.svc:8401" - - "vm-storage-datascience-prd-3.vm-storage-datascience-prd.victoriametrics.svc:8401" - - "vm-storage-datascience-prd-4.vm-storage-datascience-prd.victoriametrics.svc:8401" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-select-datascience-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # -- Pod's additional labels - podLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-select-datascience-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Readiness & Liveness probes - probe: - # -- VMSelect readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 40 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 10 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vmselect - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vmselect - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - cloud.google.com/compute-class: "vmselect" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 38 - memory: 70Gi - - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Create VMSelect service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: false - - # -- Ingress annotations - annotations: {} - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-select-datascience.prd.meesho.int - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # -- Ingress controller class name - ingressClassName: contour-internal-0 - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for select component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/select" .Values.vmselect }}' - - # -- vmselect mode: deployment, daemonSet - mode: deployment - - # -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables - deployment: - spec: - # -- VMSelect strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - - # -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables - statefulSet: - enabled: false - spec: - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume extra labels - extraLabels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmselect component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: false - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - diff --git a/helm-overrides/gke-datascience-prd-as1a/victoriametrics-storage/custom-values.yaml b/helm-overrides/gke-datascience-prd-as1a/victoriametrics-storage/custom-values.yaml deleted file mode 100644 index 9f624d7..0000000 --- a/helm-overrides/gke-datascience-prd-as1a/victoriametrics-storage/custom-values.yaml +++ /dev/null @@ -1,363 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - enabled: false - -vminsert: - enabled: false - -vmauth: - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- Override default `app` label name - name: vmstorage - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-south1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Variant of the image to use. e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMStorage http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-datascience-prd - - # -- Additional environment variables (ex.: secret tokens, flags). Check https://docs.victoriametrics.com/victoriametrics/#environment-variables for details - env: [] - # -- Specify alternative source for env variables - envFrom: [] - # -- Data retention period. Possible units character: h(ours), d(ays), w(eeks), y(ears), if no unit character specified - month. The minimum retention period is 24h. - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - httpListenAddr: :8482 - loggerTimezone: "Asia/Kolkata" - dedup.minScrapeInterval: 60s - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmstorage-n4d" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - cloud.google.com/compute-class: "vmstorage-n4d" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - # -- Use an alternate scheduler, e.g. "stork". Check https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ for details - # schedulerName: - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Override Persistent Volume Claim name - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are https://kubernetes.io/docs/concepts/storage/persistent-volumes/ - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume extra labels - extraLabels: {} - # -- Storage class name. Will be empty if not set - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume - size: 8134Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-storage-datascience-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - # -- Pod's additional labels - podLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-storage-datascience-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Details are https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 42 - memory: 330Gi - - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: - enabled: false - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - podSecurityContext: - enabled: false - service: - enabled: true - # -- Service annotations - annotations: {} - # -- Service ClusterIP - clusterIP: None - # -- Service type - type: ClusterIP - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmselect - vmselectPort: 8401 - # -- Extra service ports - extraPorts: [] - # -- Health check node port for a service. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilyPolicy: "" - # -- List of service IP families. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilies: [] - # -- Traffic Distribution. Check https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution - trafficDistribution: "" - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - minReadySeconds: 5 - # -- Readiness probes - probe: - # -- VMStorage readiness probe - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMStorage startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmstorage component - enabled: false - # -- Maximum replicas for HPA to use to to scale the vmstorage component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmstorage component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmstorage component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: - scaleDown: - selectPolicy: Disabled - - vmbackupmanager: - # -- Enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-datascience-prd-ase1a/README.md b/helm-overrides/gke-datascience-prd-ase1a/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/gke-datascience-prd-ase1a/alloy/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/alloy/custom-values.yaml deleted file mode 100644 index 68f594c..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/alloy/custom-values.yaml +++ /dev/null @@ -1,53 +0,0 @@ -fullnameOverride: "alloy-datascience-a-prd" - -alloy: - configMap: - configFile: datascience.alloy - - clustering: - enabled: true - - extraPorts: - - name: "otlp-grpc" - port: 4317 - targetPort: 4317 - protocol: "TCP" - - name: "otlp-http" - port: 4318 - targetPort: 4318 - protocol: "TCP" - - resources: - requests: - cpu: 7 - memory: 55Gi - limits: - cpu: 7.5 - memory: 60Gi - -configReloader: - enabled: true - -serviceAccount: - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-sre-grafna-obs-stk-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - -controller: - # Must be one of 'daemonset', 'deployment', or 'statefulset'. - type: 'deployment' - - nodeSelector: - cloud.google.com/compute-class: "alloy" - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "alloy" - effect: "NoSchedule" - - autoscaling: - enabled: true - minReplicas: 1 - maxReplicas: 50 - targetCPUUtilizationPercentage: 80 - targetMemoryUtilizationPercentage: 60 \ No newline at end of file diff --git a/helm-overrides/gke-datascience-prd-ase1a/aurva-dataplane/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/aurva-dataplane/custom-values.yaml deleted file mode 100644 index 893562b..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/aurva-dataplane/custom-values.yaml +++ /dev/null @@ -1,694 +0,0 @@ -## Aurva Data Plane -## Ref: https://github.com/aurva-io/aurva-charts.git - -postgresql: - enabled: true - fullnameOverride: "aurva-dataplane-database" - volumePermissions: - ## @param volumePermissions.enabled Enable init container that changes the owner and group of the persistent volume - ## - enabled: true - global: - storageClass: hyperdisk-balanced - postgresql: - auth: - postgresPassword: "aurva" - database: "controller" - # Add toleration to make sure where this postgres db pod should reside (Applicable for production workloads): For more detail ref: https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/ - primary: - extendedConfiguration: | - max_connections = 300 - - #PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - - # -- Select nodes to deploy which matches the following labels - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: datascience-devops - -# -- Provide a name in place of `aurva` -# namespaceOverride: aurva-dataplane - -########################################################## -# Global Configs -########################################################## -global: - aurva_controller: - enabled: true - aurva_fastdet: - enabled: false - aurva_pii_analyzer: - enabled: true - aurva_ocr: - enabled: true - aurva_collector: - enabled: true - - deploymentAnnotations: {} - - priorityClassName: "" - -########################################################## -# Aurva Controller -########################################################## -aurva_controller: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "datascience" - team: "datascience-devops" - service: "aurva-datascience-a-prd" - env: "prd" - priority: "p0" - type: "aurva_controller" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 10 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - image: - # Image of the app container - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-controller/aurva-controller - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - - type: secret - name: aurva-controller-secrets - # - type: configmap - # name: proxy-datasource-config - - # -- Resources to be defined for pod - resources: - limits: - memory: 2.5Gi - cpu: 2 - requests: - memory: 2Gi - cpu: 1 - - aurvaFastdet: - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-fastdet/aurva-fastdet - tag: "v2.30.13" - pullPolicy: IfNotPresent - envFrom: [] - env: [] - resources: - limits: - cpu: 0.5 - memory: 512Mi - requests: - cpu: 0.5 - memory: 512Mi - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: datascience-devops - - # -- Taint tolerations for nodes - ##PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-controller-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - config: - #variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - COMMAND_URL: "command.aurva-prd.meeshogcp.in:80" - DEPLOYMENT_TYPE: "kubernetes" - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - FLUSHER_WORKER_POOL_SIZE: "10000" - FLUSHER_BATCH_SIZE: "30000" - UNIQUENESS_IDENTIFIER: "k8s-datascience-prd-ase1" #Recommendation: should be equal to cluster name - PROVIDER_ACCOUNT_ID: "meesho-datascience-prd-0622" # GCP PROJECT ID (not Number) - REGION: "asia-southeast1" #eg: asia-south1 - ENVIRONMENT: "prod" - OCR_ENABLED: "true" - MONITORING_ENABLED: "false" - HYBRID_ONLY_MODE: "false" - FORCE_TLS: "false" - FASTDET_FLAG : "false" - AADHAAR_ENHANCER: "0" - WORKSPACE_EVENT_TRACKING_ENABLED: "false" - ACCESS_IQ_ENABLED: "true" - GRPC_ENFORCE_ALPN_ENABLED: "false" - ENABLE_FASTDET: "true" - FASTDET_MAX_BATCH_SIZE: "100" - SCAN_UUID_ENABLED: "true" - HEARTBEAT_INTERVAL: "5m" - #pii data - PII_BUCKET_NAME: "gcs-infra-devop-aurva-datascience-a-prd" - PII_LOG_BUCKET_REGION: "asia-southeast1" - PII_LOG_CRON: "*/5 * * * *" - ENABLE_PII_LOG: "true" - PII_EVIDENCE_UPLOAD_MAX_BLOCKING_TASKS: "50000" - PII_EVIDENCE_UPLOAD_MAX_CONCURRENT_TASKS: "500" - PII_EVIDENCE_MAX_CACHE_WEIGHT: "100" - QUOTA_CLEANUP_CRON: "0 0 * * *" - ENABLE_PII_QUOTA: "true" - MAX_PII_EVIDENCES_PER_KEY_PER_WINDOW: "3" - PII_QUOTA_SYNC_CRON: "*/2 * * * *" - QUOTA_WINDOW_HOURS: "24" - #constants - SKIP_NAMESPACES: "contour-internal-1-datascience-a-prd,contour-internal-1-datascience-a-prd-intra,contour-internal-0-datascience-a-prd,contour-internal-0-datascience-a-prd-intra" - CLOUD_PROVIDER: "gcp" - LOG_ENV: "production" - RDS_SCANNER_AVAILABILITY : "false" - REDSHIFT_SCANNER_AVAILABILITY : "false" - S3_SCANNER_AVAILABILITY : "false" - DYNAMO_SCANNER_AVAILABILITY: "false" - DOCDB_SCANNER_AVAILABILITY: "false" - OPENSEARCH_SCANNER_AVAILABILITY: "false" - CLOUDSQL_SCANNER_AVAILABILITY: "true" - BIGQUERY_SCANNER_AVAILABILITY: "true" - AWS_SNAPSHOT_SCANNER_AVAILABILITY: "false" - CLOUDSTORAGE_SCANNER_AVAILABILITY: "true" - KEYSPACES_SCANNER_AVAILABILITY: "false" - ALLOYDB_SCANNER_AVAILABILITY: "true" - BIGTABLE_SCANNER_AVAILABILITY: "true" - GCP_BACKUP_AVAILABILITY: "true" - EGRESS_MODE_ONLY: "false" - SCAN_UUID_ENABLED : "true" - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-controller-sa - # -- Annotations applied to created service account - annotations: - iam.gke.io/gcp-service-account: sa-datascience-prd-aurva-contr@meesho-datascience-prd-0622.iam.gserviceaccount.com -# eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 15 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 - - -########################################################## -# Aurva OCR -########################################################## -aurva_ocr: - # -- Additional labels for aurva-controller - additionalLabels: - bu: "datascience" - team: "datascience-devops" - service: "aurva-datascience-a-prd" - env: "prd" - priority: "p0" - type: "aurva_ocr" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 1 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-ocr/aurva-ocr - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-ocr: - type: secret - name: aurva-ocr-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 1 - requests: - memory: 2Gi - cpu: 1 - - # -- Select nodes to deploy which matches the following labels - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: datascience-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-ocr-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - OCR_TIME_LIMIT: "1" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "gke-datascience-prd-ase1a" - DEPLOYMENT_TYPE: "kubernetes" - MAX_HOURS_TO_KEEP_REGARDLESS_OF_ACK_STATUS : "2880" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-ocr-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # iam.gke.io/gcp-service-account: service-account@gcp.iam.gserviceaccount.com - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva Collector -########################################################## -aurva_collector: - - # -- Additional labels for aurva-analyzer - additionalLabels: - bu: "datascience" - team: "datascience-devops" - service: "aurva-datascience-a-prd" - env: "prd" - priority: "p0" - type: "aurva_collector" - - - # -- Annotations on aurva-analyzer - annotations: {} - # "key": "value" - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-collector/aurva-collector - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-controller: - type: secret - name: aurva-collector-secrets - - resources: - limits: - cpu: 200m - memory: 800Mi - requests: - cpu: 200m - memory: 512Mi - - podSecurityContext: {} - - securityContext: - privileged: true - capabilities: - add: - # For kernel v5.8 and above we don't need CAP_SYS_ADMIN or CAP_SYS_RESOURCE - # we just need CAP_BPF and CAP_PERFMON. This has been tested on our EKS node - # which is on kernel v5.10.x - # When SSL Tracing is required we need CAP_SYS_ADMIN and CAP_SYS_PTRACE - # on top of the previous capabilities - # So finally these are the 4 possible combinations for capabilies - # 1. Newer Kernels without SSL - # - BPF - # - PERFMON - # 2. Newer Kernels with SSL - - SYS_ADMIN - - SYS_PTRACE - # 3. Older Kernels without SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # 4. Older Kernels with SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # - SYS_PTRACE - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - volumes: - - name: debugfs - mountPath: /sys/kernel/debug - hostPath: /sys/kernel/debug - - name: vmlinux - mountPath: /sys/kernel/btf/vmlinux - hostPath: /sys/kernel/btf/vmlinux - - name: procfs - mountPath: /host/proc - hostPath: /proc - - name: bpffs - mountPath: /sys/fs/bpf - hostPath: /sys/fs/bpf - - # -- Taint tolerations for nodes - tolerations: - # - effect: NoSchedule - # key: cloud.google.com/compute-class - # operator: Equal - # value: megatetra - - operator: Exists - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vm-stack-ht - - vmagent-c4 - - vmagent - - vmagent-mds - - vmagent-n4 - - vminsert - - vminsert-mds - - vmselect - - vmselect-mds - - vmstorage-c4d - - vmstorage - - vmstorage-mds - - vmstorage-n4 - - vmstorage-n4d - - vmstorage-sale-24aug - - vmstorage-tmp - - contour-external - - contour-internal-0 - - contour-internal-1 - - contour-internal-1-dataproc-cc - - contour-intra-0 - - contour-intra-1 - - contour-shared - - alloy - - preprod-spot-16 - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-collector-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - # variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "gke-datascience-prd-ase1a" - DEPLOYMENT_TYPE: "kubernetes" - TRACE_INTERNAL_SVC: "true" - TRACE_INTERNAL_SVC_HTTP: "true" - INTERNAL_SVC_SAMPLE_INTERVAL: "10m" - LOGS_TTL: "1h" - TRACE_HTTP2: "true" - TRACE_SSL: "false" - TRACE_PSQL: "false" - TRACE_SQLSERVER: "false" - TRACE_MYSQL: "false" - TRACE_EGRESS: "true" - TRACE_GO_TLS: "false" - TRACE_ML_SERVICES: "false" - # constants - LOG_ENV: production - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - MONITORING_ENABLED: "false" - ENABLE_INGRESS_INFORMER: "false" - ENABLE_SERVICE_INFORMER: "false" - ENABLE_ISTIO_INFORMER: "false" - EXCLUDED_PII_REGEX_TYPES: "ip_address,us_bank_number,us_driver_license,us_itin,us_passport,us_routing,us_mbi,ssn" - AGGREGATOR_MAX_CONNECTIONS: "1000" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-collector-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva PII Analyzer -########################################################## -aurva_pii_analyzer: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "datascience" - team: "datascience-devops" - service: "aurva-datascience-a-prd" - env: "prd" - priority: "p0" - type: "aurva_pii_analyzer" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 16 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - ##PLACEHOLDER## - nodeSelector: - cloud.google.com/compute-class: datascience-devops - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-piianalyzer/aurva-piianalyzer - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-pii-analyzer: - type: secret - name: aurva-pii-analyzer-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 5Gi - cpu: 4 - requests: - memory: 4Gi - cpu: 2 - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: datascience-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-pii-analyzer-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - SCHEDULER_TIME: "1" - SUPPORTED_REGION: "US" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "gke-datascience-prd-ase1a" - DEPLOYMENT_TYPE: "kubernetes" - MAX_HOURS_TO_KEEP_REGARDLESS_OF_ACK_STATUS: "2880" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-pii-analyzer-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - - autoscaling: - enabled: true - minReplicas: 1 - maxReplicas: 2 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 diff --git a/helm-overrides/gke-datascience-prd-ase1a/cert-manager/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/cert-manager/custom-values.yaml deleted file mode 100644 index 124a27c..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/cert-manager/custom-values.yaml +++ /dev/null @@ -1,129 +0,0 @@ -global: - logLevel: 2 - rbac: - create: true - priorityClassName: "high-priority" -installCRDs: false - -crds: - enabled: true - keep: true - -# Cert-manager Controller -replicaCount: 1 -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-controller - tag: v1.20.1 - pullPolicy: IfNotPresent - -nodeSelector: - cloud.google.com/compute-class: datascience-devops - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - -resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 500m - memory: 512Mi - -# Webhook Configuration -webhook: - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-webhook - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: datascience-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# CA Injector Configuration -cainjector: - enabled: true - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-cainjector - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: datascience-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# ACME Solver Configuration -acmesolver: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-acmesolver - tag: v1.20.1 - pullPolicy: IfNotPresent - -# Startup API Check -startupapicheck: - enabled: true - timeout: 1m - backoffLimit: 4 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-startupapicheck - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: datascience-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - -# Prometheus Monitoring -prometheus: - enabled: true - servicemonitor: - enabled: false - interval: 60s - scrapeTimeout: 30s - labels: - prometheus: cert-manager diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/c3-standard-22-lssd-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/c3-standard-22-lssd-cc.yaml deleted file mode 100644 index 710279a..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/c3-standard-22-lssd-cc.yaml +++ /dev/null @@ -1,23 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: c3-standard-22-lssd -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: c3-standard-22-lssd - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-standard-22-lssd - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - localSSDCount: 4 - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-internal-0-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-internal-0-cc.yaml deleted file mode 100644 index cbc81d3..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-internal-0-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-0 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-internal-0 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - - machineType: c3-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-internal-1-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-internal-1-cc.yaml deleted file mode 100644 index 4e07d8b..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-internal-1-cc.yaml +++ /dev/null @@ -1,34 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-internal-1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - - machineType: c3-highcpu-22 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-internal-dataproc-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-internal-dataproc-cc.yaml deleted file mode 100644 index a7a1e03..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-internal-dataproc-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-dataproc -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-internal-dataproc - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - - machineType: c3-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-intra-0-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-intra-0-cc.yaml deleted file mode 100644 index c02a5c7..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-intra-0-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-0 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-intra-0 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - - machineType: c3-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-intra-1-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-intra-1-cc.yaml deleted file mode 100644 index 3d55d6a..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-intra-1-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-intra-1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - - machineType: c3-highcpu-22 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-shared-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-shared-cc.yaml deleted file mode 100644 index 56bec9c..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/contour-shared-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-shared -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-shared - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: n4-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/datascience-devops-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/datascience-devops-cc.yaml deleted file mode 100644 index 572d8f0..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/datascience-devops-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: datascience-devops -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: datascience-devops - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: e2-standard-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/datascience-kyverno-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/datascience-kyverno-cc.yaml deleted file mode 100644 index 5f0dca0..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/datascience-kyverno-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: datascience-kyverno -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: datascience-kyverno - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-standard-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/ds-airflow-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/ds-airflow-cc.yaml deleted file mode 100644 index 2bc7ee5..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/ds-airflow-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: ds-airflow -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: ds-airflow - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-standard-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/g2-l4-priority-class-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/g2-l4-priority-class-cc.yaml deleted file mode 100644 index 27b9cc3..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/g2-l4-priority-class-cc.yaml +++ /dev/null @@ -1,26 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-l4-priority-class -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: g2-l4-priority-class - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - machineType: g2-standard-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/g2-standard-4-l4-compute-class-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/g2-standard-4-l4-compute-class-cc.yaml deleted file mode 100644 index 2c4862d..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/g2-standard-4-l4-compute-class-cc.yaml +++ /dev/null @@ -1,26 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-4-l4-compute-class -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: g2-standard-4-l4-compute-class - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - machineType: g2-standard-4 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/g2-standard-8-l4-300gb-compute-class-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/g2-standard-8-l4-300gb-compute-class-cc.yaml deleted file mode 100644 index fcdba76..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/g2-standard-8-l4-300gb-compute-class-cc.yaml +++ /dev/null @@ -1,26 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-8-l4-300gb-compute-class -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: g2-standard-8-l4-300gb-compute-class - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - machineType: g2-standard-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 300 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/g2-standard-8-l4-compute-class-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/g2-standard-8-l4-compute-class-cc.yaml deleted file mode 100644 index 3181f74..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/g2-standard-8-l4-compute-class-cc.yaml +++ /dev/null @@ -1,54 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-8-l4-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: g2-standard-8-l4-compute-class - priorities: - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-a - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-b - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-c - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/load-testing-v2-mlp-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/load-testing-v2-mlp-cc.yaml deleted file mode 100644 index ac5de12..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/load-testing-v2-mlp-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: load-testing-v2-mlp -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: load-testing-v2-mlp - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-standard-48 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/megaduo-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/megaduo-cc.yaml deleted file mode 100644 index 652f15b..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/megaduo-cc.yaml +++ /dev/null @@ -1,34 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaduo -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megaduo - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-highcpu-22 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: c3-highcpu-44 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: c4-highcpu-48 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/megaduo-ctz-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/megaduo-ctz-cc.yaml deleted file mode 100644 index 16991e3..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/megaduo-ctz-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaduo-ctz -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megaduo-ctz - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-highcpu-22 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/megaduo-rto-consumer-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/megaduo-rto-consumer-cc.yaml deleted file mode 100644 index 3910f64..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/megaduo-rto-consumer-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaduo-rto-consumer -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megaduo-rto-consumer - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-highcpu-22 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/megaduolite-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/megaduolite-cc.yaml deleted file mode 100644 index 9261a4f..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/megaduolite-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaduolite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megaduolite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-custom-16-32768 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/megaquad-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/megaquad-cc.yaml deleted file mode 100644 index d0fd2be..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/megaquad-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaquad -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megaquad - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-standard-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/megatetra-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/megatetra-cc.yaml deleted file mode 100644 index 41968d4..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/megatetra-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megatetra -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megatetra - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-standard-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/megatetralite-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/megatetralite-cc.yaml deleted file mode 100644 index d0aad81..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/megatetralite-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megatetralite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megatetralite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-standard-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-a2-highgpu-1-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-a2-highgpu-1-cc.yaml deleted file mode 100644 index d20dcf7..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-a2-highgpu-1-cc.yaml +++ /dev/null @@ -1,26 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: mlp-a2-highgpu-1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: mlp-a2-highgpu-1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-tesla-a100 - machineType: a2-highgpu-1g - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-c3d-hc-30-300-v1-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-c3d-hc-30-300-v1-cc.yaml deleted file mode 100644 index 3f0e838..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-c3d-hc-30-300-v1-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: mlp-c3d-hc-30-300-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: mlp-c3d-hc-30-300-v1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3d-highcpu-30 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 300 - bootDiskType: pd-ssd - - machineType: c4d-standard-32 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 300 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-c3d-highcpu-30-sz-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-c3d-highcpu-30-sz-cc.yaml deleted file mode 100644 index 2ab6102..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-c3d-highcpu-30-sz-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: mlp-c3d-highcpu-30-sz -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: mlp-c3d-highcpu-30-sz - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3d-highcpu-30 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-c3d-highcpu-30-v1-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-c3d-highcpu-30-v1-cc.yaml deleted file mode 100644 index 54ad602..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-c3d-highcpu-30-v1-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: mlp-c3d-highcpu-30-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: mlp-c3d-highcpu-30-v1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3d-highcpu-30 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-c4d-localssd-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-c4d-localssd-cc.yaml deleted file mode 100644 index 341ecf4..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-c4d-localssd-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: mlp-c4d-localssd -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: mlp-c4d-localssd - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4d-standard-16-lssd - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-g2-standard-16-v2-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-g2-standard-16-v2-cc.yaml deleted file mode 100644 index 4c8d866..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-g2-standard-16-v2-cc.yaml +++ /dev/null @@ -1,26 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: mlp-g2-standard-16-v2 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: mlp-g2-standard-16-v2 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - machineType: g2-standard-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-g2-standard-32-custom-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-g2-standard-32-custom-cc.yaml deleted file mode 100644 index ec6f816..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-g2-standard-32-custom-cc.yaml +++ /dev/null @@ -1,26 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: mlp-g2-standard-32-custom -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: mlp-g2-standard-32-custom - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - machineType: g2-custom-32-98304 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-g2-standard-8-zone-a-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-g2-standard-8-zone-a-cc.yaml deleted file mode 100644 index 733ac75..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/mlp-g2-standard-8-zone-a-cc.yaml +++ /dev/null @@ -1,26 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: mlp-g2-standard-8-zone-a -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: mlp-g2-standard-8-zone-a - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - machineType: g2-standard-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/n2d-standard-48-4lssd-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/n2d-standard-48-4lssd-cc.yaml deleted file mode 100644 index 30676c0..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/n2d-standard-48-4lssd-cc.yaml +++ /dev/null @@ -1,23 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: n2d-standard-48-4lssd -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: n2d-standard-48-4lssd - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-standard-48 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - localSSDCount: 4 - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/nginx-internal-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/nginx-internal-cc.yaml deleted file mode 100644 index ca5aff5..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/nginx-internal-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: nginx-internal -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: nginx-internal - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-highcpu-4 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/np-dsci-ml-g2-standard-8-prd-ase1-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/np-dsci-ml-g2-standard-8-prd-ase1-cc.yaml deleted file mode 100644 index 01b9a1b..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/np-dsci-ml-g2-standard-8-prd-ase1-cc.yaml +++ /dev/null @@ -1,26 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: np-dsci-ml-g2-standard-8-prd-ase1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: np-dsci-ml-g2-standard-8-prd-ase1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - machineType: g2-standard-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/rockdb-localssd-1-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/rockdb-localssd-1-cc.yaml deleted file mode 100644 index dc0a87c..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/rockdb-localssd-1-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: rockdb-localssd-1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: rockdb-localssd-1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4-standard-16-lssd - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/rust-onboard-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/rust-onboard-cc.yaml deleted file mode 100644 index c80082c..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/rust-onboard-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: rust-onboard -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: rust-onboard - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4a-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/ssd-cosmos-v2-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/ssd-cosmos-v2-cc.yaml deleted file mode 100644 index ae789dd..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/ssd-cosmos-v2-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: ssd-cosmos-v2 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: ssd-cosmos-v2 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-standard-32 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/sumoduo-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/sumoduo-cc.yaml deleted file mode 100644 index d3cca03..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/sumoduo-cc.yaml +++ /dev/null @@ -1,34 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumoduo -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumoduo - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-highcpu-44 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: c3-highcpu-88 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: c4-highcpu-96 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/sumotetra-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/sumotetra-cc.yaml deleted file mode 100644 index 13e9fad..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/sumotetra-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumotetra -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumotetra - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-standard-32 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmagent-dr-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmagent-dr-cc.yaml deleted file mode 100644 index 7fd0a3c..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmagent-dr-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmagent-dr -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmagent-dr - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: TODO-vmagent-dr - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmagent-mds-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmagent-mds-cc.yaml deleted file mode 100644 index 11b692b..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmagent-mds-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmagent-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmagent-mds - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: TODO-vmagent-mds - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmagent-n4-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmagent-n4-cc.yaml deleted file mode 100644 index 44cd43c..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmagent-n4-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmagent-n4 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmagent-n4 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-80 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmagent-n4d-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmagent-n4d-cc.yaml deleted file mode 100644 index af94264..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmagent-n4d-cc.yaml +++ /dev/null @@ -1,34 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmagent-n4d -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmagent-n4d - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-80 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: c4-highcpu-192 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-384 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vminsert-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/vminsert-cc.yaml deleted file mode 100644 index 598552a..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vminsert-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vminsert -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vminsert - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: TODO-vminsert - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vminsert-mds-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/vminsert-mds-cc.yaml deleted file mode 100644 index 92f6a2b..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vminsert-mds-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vminsert-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vminsert-mds - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmselect-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmselect-cc.yaml deleted file mode 100644 index bdeba22..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmselect-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmselect -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmselect - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-highcpu-44 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmselect-mds-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmselect-mds-cc.yaml deleted file mode 100644 index f371980..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmselect-mds-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmselect-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmselect-mds - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: TODO-vmselect-mds - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmstorage-c4d-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmstorage-c4d-cc.yaml deleted file mode 100644 index f53010c..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmstorage-c4d-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage-c4d -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmstorage-c4d - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: TODO-vmstorage-c4d - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmstorage-n4-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmstorage-n4-cc.yaml deleted file mode 100644 index 67a86c2..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmstorage-n4-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage-n4 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmstorage-n4 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-highmem-48 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml deleted file mode 100644 index 4f0ecc7..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage-n4d -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmstorage-n4d - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highmem-48 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n4d-highmem-64 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmstorage-sale-24aug-cc.yaml b/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmstorage-sale-24aug-cc.yaml deleted file mode 100644 index 35ef8b7..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/computeclass/vmstorage-sale-24aug-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage-sale-24aug -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dsci-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmstorage-sale-24aug - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: TODO-vmstorage-sale-24aug - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-datascience-prd-ase1a/conntrack-adjuster/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 4732be2..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,20 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - max: 2097152 - hashsize: 524288 - sleepInterval: 30 - -nodeAffinity: - enabled: true - nodeSelectorTerms: - - matchExpressions: - - key: cloud.google.com/compute-class - operator: In - values: - - "contour-external" - - "contour-internal-0" - - "contour-internal-1" - - "contour-intra-0" - - "contour-intra-1" diff --git a/helm-overrides/gke-datascience-prd-ase1a/contour-ca-issuer/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/contour-ca-issuer/custom-values.yaml deleted file mode 100644 index 5a14106..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/contour-ca-issuer/custom-values.yaml +++ /dev/null @@ -1,17 +0,0 @@ -# ClusterIssuer override for gke-datascience-prd-ase1a (prd datascience cluster, ase1a zone). -# ClusterIssuer is a cluster-scoped resource — this file pins the issuer -# identity for this cluster so the name is auditable per-cluster. -# -# Must match the `issuerRef.name` in consuming Certificate CRs (see -# devops-helm-charts/2.0.0/templates/proxyless-grpc-cert.yaml). - -issuerName: contour-datascience-a-prd-ca-issuer -rootCASecretName: contour-datascience-ca - -externalSecret: - enabled: true - vaultPath: meesho/devops/contour/root-ca - refreshInterval: "0" - secretStoreRef: - name: vault-backend -namespace: cert-manager-datascience-prd diff --git a/helm-overrides/gke-datascience-prd-ase1a/contour-cert-checker/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/contour-cert-checker/custom-values.yaml deleted file mode 100644 index e899d47..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=gke-datascience-prd-ase1a"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - cloud.google.com/compute-class: datascience-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/gke-datascience-prd-ase1a/contour-internal-0/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/contour-internal-0/custom-values.yaml deleted file mode 100644 index 03e4258..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,127 +0,0 @@ -fullnameOverride: "contour-internal-0-datascience-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 4Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared - nodeSelector: - cloud.google.com/compute-class: contour-shared - service: - tcpLB: true - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-0 - nodeSelector: - cloud.google.com/compute-class: contour-internal-0 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "55" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 6' - resources: - requests: - cpu: 6 - memory: 8Gi - limits: - cpu: 6 - memory: 14Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-datascience-a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-datascience-prd-ase1a/contour-internal-1/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/contour-internal-1/custom-values.yaml deleted file mode 100644 index 65780a6..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,118 +0,0 @@ -fullnameOverride: "contour-internal-1-datascience-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 12 - memory: 22Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared - nodeSelector: - cloud.google.com/compute-class: contour-shared - service: - tcpLB: true - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-1 - nodeSelector: - cloud.google.com/compute-class: contour-internal-1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 14' - resources: - requests: - cpu: 14 - memory: 6Gi - limits: - cpu: 14 - memory: 27Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int1-datascience-a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-datascience-prd-ase1a/contour-internal-dataproc/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/contour-internal-dataproc/custom-values.yaml deleted file mode 100644 index 6713b98..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/contour-internal-dataproc/custom-values.yaml +++ /dev/null @@ -1,112 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 60s - max-connection-duration: 600s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 500m - memory: 3Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared - nodeSelector: - cloud.google.com/compute-class: contour-shared - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-dataproc" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 5% - maxUnavailable: 0 - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-dataproc - nodeSelector: - cloud.google.com/compute-class: contour-internal-dataproc - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 6' - resources: - requests: - cpu: 6 - memory: 8Gi - limits: - cpu: 6 - memory: 14Gi - service: - tcpLB: true - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/gke-datascience-prd-ase1a/contour-internal-intra-0/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index 3371a1f..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,125 +0,0 @@ -fullnameOverride: "contour-internal-0-datascience-intra-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 4Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared - nodeSelector: - cloud.google.com/compute-class: contour-shared - service: - tcpLB: true - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-intra-0 - nodeSelector: - cloud.google.com/compute-class: contour-intra-0 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "55" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 6' - resources: - requests: - cpu: 6 - memory: 8Gi - limits: - cpu: 6 - memory: 14Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-datascience-prd-ase1a/contour-internal-intra-1/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index 2daa589..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -fullnameOverride: "contour-internal-1-datascience-intra-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 4Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared - nodeSelector: - cloud.google.com/compute-class: contour-shared - service: - tcpLB: false - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-intra-1 - nodeSelector: - cloud.google.com/compute-class: contour-intra-1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 10' - resources: - requests: - cpu: 10 - memory: 20Gi - limits: - cpu: 14 - memory: 30Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-datascience-prd-ase1a/coredns/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/coredns/custom-values.yaml deleted file mode 100644 index b45662a..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/coredns/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -replicaCount: 30 -communicationType: "intra" - -labels: - bu: datascience - team: datascience-devops - env: prd - -clusterIP: 10.1.48.2 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - -nodeSelector: - cloud.google.com/compute-class: datascience-devops - kubernetes.io/os: linux diff --git a/helm-overrides/gke-datascience-prd-ase1a/coroot-node-agent/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/coroot-node-agent/custom-values.yaml deleted file mode 100644 index bae0859..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/coroot-node-agent/custom-values.yaml +++ /dev/null @@ -1,54 +0,0 @@ -fullnameOverride: "coroot-node-agent-datascience-a-prd" -priorityClassName: "low-priority" - -scrape: "false" # disable scraping for now - -resources: - requests: - cpu: "300m" - memory: "1Gi" - limits: - cpu: "500m" - memory: "1.5Gi" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-datascience-contour-ext - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-datascience-default-prd-ase1 - - key: cloud.google.com/compute-class - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external - - alloy - - sumoduolite-op - - sumounolite \ No newline at end of file diff --git a/helm-overrides/gke-datascience-prd-ase1a/external-secrets/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/external-secrets/custom-values.yaml deleted file mode 100644 index 29867ec..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/external-secrets/custom-values.yaml +++ /dev/null @@ -1,61 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Annotations to add to Pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - nodeSelector: - cloud.google.com/compute-class: datascience-devops - certController: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - nodeSelector: - cloud.google.com/compute-class: datascience-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - webhook: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - nodeSelector: - cloud.google.com/compute-class: datascience-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" \ No newline at end of file diff --git a/helm-overrides/gke-datascience-prd-ase1a/flagger/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/flagger/custom-values.yaml deleted file mode 100644 index 3392ce4..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/flagger/custom-values.yaml +++ /dev/null @@ -1,63 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - cloud.google.com/compute-class: datascience-devops - -tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: datascience-devops - bu: infra - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-datascience-rollout-service.prd-datascience-rollout-service.svc.cluster.local" - -rolloutService: - enabled: true - apps: - \ No newline at end of file diff --git a/helm-overrides/gke-datascience-prd-ase1a/fluentd/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/fluentd/custom-values.yaml deleted file mode 100644 index 6592b3c..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/fluentd/custom-values.yaml +++ /dev/null @@ -1,740 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-dssre-fluentd-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: cloud.google.com/compute-class - operator: NotIn - values: - - no-exclusion -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: datascience - team: sre - type: fluentd - service: fluentd-datascience-a-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: datascience - team: sre - type: fluentd - service: fluentd-datascience-a-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -# - secretRef: -# name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: - - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - from_encoding "utf-8" - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/gke-datascience-prd-ase1a/ingress-nginx-internal/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/ingress-nginx-internal/custom-values.yaml deleted file mode 100644 index 9279958..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/ingress-nginx-internal/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 6 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dsci-int-a-prd"}}}' - nodeSelector: - cloud.google.com/compute-class: nginx-internal - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "nginx-internal" - effect: "NoSchedule" diff --git a/helm-overrides/gke-datascience-prd-ase1a/keda/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/keda/custom-values.yaml deleted file mode 100644 index a915fd5..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/keda/custom-values.yaml +++ /dev/null @@ -1,45 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - cloud.google.com/compute-class: datascience-devops - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - podLabels: - bu: "datascience" - team: "datascience-devops" - metricsAdapter: - bu: "datascience" - team: "datascience-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 2000Mi - requests: - cpu: 700m - memory: 1000Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 120m - memory: 300Mi - metricServer: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 120m - memory: 200Mi - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' diff --git a/helm-overrides/gke-datascience-prd-ase1a/kube-dns/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/kube-dns/custom-values.yaml deleted file mode 100644 index 8f6b085..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.1.48.2"],"prd.mrouter.int.svc.cluster.local":["10.1.16.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.1.16.2"]} diff --git a/helm-overrides/gke-datascience-prd-ase1a/kube-events/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/kube-events/custom-values.yaml deleted file mode 100644 index 1f5eca2..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-datascience-a-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - cloud.google.com/compute-class: datascience-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: datascience-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: datascience-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: datascience-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: datascience-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: datascience-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/gke-datascience-prd-ase1a/kube-state-metrics/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/kube-state-metrics/custom-values.yaml deleted file mode 100644 index cff3d23..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-datascience-a-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-datascience-a-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "datascience" - team: "datascience-sre" - service: "kube-state-metrics-datascience-a-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - cloud.google.com/compute-class: "datascience-devops" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 600m - memory: 1.5Gi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/gke-datascience-prd-ase1a/kubectl-mcp-server/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/kubectl-mcp-server/custom-values.yaml deleted file mode 100644 index c04817b..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/kubectl-mcp-server/custom-values.yaml +++ /dev/null @@ -1,117 +0,0 @@ -fullnameOverride: "kubectl-mcp-server" - -replicas: 1 - -image: - # In-house hardened build (helm-templates/kubectl-mcp-server/docker/) — - # NOT the upstream Docker Hub image. 266→18 HIGH/CRIT vulns, non-root, - # multi-stage slim base, kubectl v1.33.12 / helm v3.21.0. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/prd/devop/kubectl-mcp-server - tag: "v2" - pullPolicy: IfNotPresent - -labels: - bu: infra - team: devops - service: kubectl-mcp-server - env: prd - -serviceAccount: - create: true - annotations: {} - -rbac: - create: true - -podAnnotations: {} - -podSecurityContext: {} - -securityContext: - runAsNonRoot: true - runAsUser: 1000 - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -priorityClassName: "" - -nodeSelector: - cloud.google.com/compute-class: "datascience-devops" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: NoSchedule - -affinity: {} - -resources: - requests: - cpu: 250m - memory: 512Mi - limits: - cpu: 1000m - memory: 1Gi - -mcp: - mode: single - transport: http - host: "0.0.0.0" - port: 8000 - -auth: - allowAnonymous: false - -externalSecrets: - enabled: true - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/prd/cntr/devop/kubectl-mcp-server-datascience" - -# tcpSocket on purpose — streamable-http transport exposes only /mcp, -# there is no /health route (see chart values.yaml note). -livenessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - -readinessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - -service: - port: 8000 - type: ClusterIP - -# Contour standard (2.0.0-style): renders HTTPProxy parent/child + -intra -# instead of a networking.k8s.io Ingress. -createContourGateway: true - -ingress: - enabled: true - ingressClassName: contour-internal-1 - servicePortNumber: 8000 - hosts: - - host: kubectl-mcp-server-datascience.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 diff --git a/helm-overrides/gke-datascience-prd-ase1a/kubernetes-dashboard/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/kubernetes-dashboard/custom-values.yaml deleted file mode 100644 index 670d841..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/kubernetes-dashboard/custom-values.yaml +++ /dev/null @@ -1,442 +0,0 @@ -# Copyright 2017 The Kubernetes Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -# General configuration shared across resources -app: - # Mode determines if chart should deploy a full Dashboard with all containers or just the API. - # - dashboard - deploys all the containers - # - api - deploys just the API - mode: 'dashboard' - image: - pullPolicy: IfNotPresent - pullSecrets: [] - scheduling: - # Node labels for pod assignment - # Ref: https://kubernetes.io/docs/user-guide/node-selection/ - nodeSelector: {} - security: - # Allow overriding csrfKey used by API/Auth containers. - # It has to be base64 encoded random 256 bytes string. - # If empty, it will be autogenerated. - csrfKey: ~ - # SecurityContext to be added to pods - # To disable set the following configuration to null: - # securityContext: null - securityContext: - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - # ContainerSecurityContext to be added to containers - # To disable set the following configuration to null: - # containerSecurityContext: null - containerSecurityContext: - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - runAsUser: 1001 - runAsGroup: 2001 - capabilities: - drop: ["ALL"] - # Pod Disruption Budget configuration - # Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - minAvailable: 0 - maxUnavailable: 0 - networkPolicy: - enabled: false - ingressDenyAll: false - # Raw network policy spec that overrides predefined spec - # Example: - # spec: - # egress: - # - ports: - # - port: 123 - spec: {} - - # Common labels & annotations shared across all deployed resources - labels: {} - annotations: {} - # Common priority class used for all deployed resources - priorityClassName: null - settings: - ## Global dashboard settings - global: - # # Cluster name that appears in the browser window title if it is set - # clusterName: "" - # # Max number of items that can be displayed on each list page - # itemsPerPage: 10 - # # Max number of labels that are displayed by default on most views. - # labelsLimit: 3 - # # Number of seconds between every auto-refresh of logs - # logsAutoRefreshTimeInterval: 5 - # # Number of seconds between every auto-refresh of every resource. Set 0 to disable - # resourceAutoRefreshTimeInterval: 10 - # # Hide all access denied warnings in the notification panel - # disableAccessDeniedNotifications: false - # # Hide all namespaces option in namespace selection dropdown to avoid accidental selection in large clusters thus preventing OOM errors - # hideAllNamespaces: false - # # Namespace that should be selected by default after logging in. - # defaultNamespace: default - # # List of namespaces that should be presented to user without namespace list privileges. - # namespaceFallbackList: - # - default - ## Pinned resources that will be displayed in dashboard's menu - pinnedResources: [] - # - kind: customresourcedefinition - # # Fully qualified name of a CRD - # name: prometheus.monitoring.coreos.com - # # Display name - # displayName: Prometheus - # # Is this CRD namespaced? - # namespaced: true - ingress: - enabled: false - hosts: - # Keep 'localhost' host only if you want to access Dashboard using 'kubectl port-forward ...' on: - # https://localhost:8443 - - localhost - # - kubernetes.dashboard.domain.com - ingressClassName: internal-nginx - # Use only if your ingress controllers support default ingress classes. - # If set to true ingressClassName will be ignored and not added to the Ingress resources. - # It should fall back to using IngressClass marked as the default. - useDefaultIngressClass: false - # This will append our Ingress with annotations required by our default configuration. - # nginx.ingress.kubernetes.io/backend-protocol: "HTTPS" - # nginx.ingress.kubernetes.io/ssl-passthrough: "true" - # nginx.ingress.kubernetes.io/ssl-redirect: "true" - useDefaultAnnotations: true - pathType: ImplementationSpecific - # If path is not the default (/), rewrite-target annotation will be added to the Ingress. - # It allows serving Kubernetes Dashboard on a sub-path. Make sure that the configured path - # does not conflict with gateway route configuration. - path: / - issuer: - name: selfsigned - # Scope determines what kind of issuer annotation will be used on ingress resource - # - default - adds 'cert-manager.io/issuer' - # - cluster - adds 'cert-manager.io/cluster-issuer' - # - disabled - disables cert-manager annotations - scope: default - tls: - enabled: true - # If provided it will override autogenerated secret name - secretName: "" - labels: {} - annotations: {} - # Use the following toleration if Dashboard can be deployed on a tainted control-plane nodes - # - key: node-role.kubernetes.io/control-plane - # effect: NoSchedule - tolerations: [] - affinity: {} - -auth: - role: auth - image: - repository: docker.io/kubernetesui/dashboard-auth - tag: 1.2.2 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: auth - containerPort: 8000 - protocol: TCP - args: [] - env: [] - volumeMounts: - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for Auth related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -# API deployment configuration -api: - role: api - image: - repository: docker.io/kubernetesui/dashboard-api - tag: 1.10.1 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: api - containerPort: 8000 - protocol: TCP - # Additional container arguments - # Full list of arguments: https://github.com/kubernetes/dashboard/blob/master/docs/common/arguments.md - # args: - # - --system-banner="Welcome to the Kubernetes Dashboard" - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store exec logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for API related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -# WEB UI deployment configuration -web: - role: web - image: - repository: docker.io/kubernetesui/dashboard-web - tag: 1.6.0 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: web - containerPort: 8000 - protocol: TCP - # Additional container arguments - # Full list of arguments: https://github.com/kubernetes/dashboard/blob/master/docs/common/arguments.md - # args: - # - --system-banner="Welcome to the Kubernetes Dashboard" - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for WEB UI related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -### Metrics Scraper -### Container to scrape, store, and retrieve a window of time from the Metrics Server. -### refs: https://github.com/kubernetes/dashboard/tree/master/modules/metrics-scraper -metricsScraper: - enabled: true - role: metrics-scraper - image: - repository: docker.io/kubernetesui/dashboard-metrics-scraper - tag: 1.2.1 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - containerPort: 8000 - protocol: TCP - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - livenessProbe: - httpGet: - scheme: HTTP - path: / - port: 8000 - initialDelaySeconds: 30 - timeoutSeconds: 30 - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for Metrics Scraper related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -## Optional Metrics Server sub-chart configuration -## Enable this if you don't already have metrics-server enabled on your cluster and -## want to use it with dashboard metrics-scraper -## refs: -## - https://github.com/kubernetes-sigs/metrics-server -## - https://github.com/kubernetes-sigs/metrics-server/tree/master/charts/metrics-server -metrics-server: - enabled: false - args: - - --kubelet-preferred-address-types=InternalIP - - --kubelet-insecure-tls - -## Required Kong sub-chart with DBless configuration to act as a gateway -## for our all containers. -kong: - enabled: true - admin: - tls: - enabled: false - ## Configuration reference: https://docs.konghq.com/gateway/3.6.x/reference/configuration - env: - dns_order: LAST,A,CNAME,AAAA,SRV - plugins: 'off' - nginx_worker_processes: 1 - ingressController: - enabled: false - manager: - enabled: false - dblessConfig: - configMap: kong-dbless-config - proxy: - type: ClusterIP - http: - enabled: true - -## Optional Cert Manager sub-chart configuration -## Enable this if you don't already have cert-manager enabled on your cluster. -cert-manager: - enabled: false - installCRDs: true - -## Optional Nginx Ingress sub-chart configuration -## Enable this if you don't already have nginx-ingress enabled on your cluster. -nginx: - enabled: false - controller: - electionID: ingress-controller-leader - ingressClassResource: - name: internal-nginx - default: false - controllerValue: k8s.io/internal-ingress-nginx - service: - type: ClusterIP - -## Extra configurations: -## - manifests -## - predefined roles -## - prometheus -## - etc... -extras: - # Extra Kubernetes manifests to be deployed - # manifests: - # - apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: additional-configmap - # data: - # mykey: myvalue - manifests: [] - serviceMonitor: - # Whether to create a Prometheus Operator service monitor. - enabled: false - # Here labels can be added to the serviceMonitor - labels: {} - # Here annotations can be added to the serviceMonitor - annotations: {} - # metrics.serviceMonitor.metricRelabelings Specify Metric Relabelings to add to the scrape endpoint - # ref: https://github.com/coreos/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # metrics.serviceMonitor.relabelings [array] Prometheus relabeling rules - relabelings: [] - # ServiceMonitor connection scheme. Defaults to HTTPS. - scheme: https - # ServiceMonitor connection tlsConfig. Defaults to {insecureSkipVerify:true}. - tlsConfig: - insecureSkipVerify: true diff --git a/helm-overrides/gke-datascience-prd-ase1a/kyverno/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/kyverno/custom-values.yaml deleted file mode 100644 index d97afe8..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/kyverno/custom-values.yaml +++ /dev/null @@ -1,2244 +0,0 @@ -global: - - # -- Internal settings used with `helm template` to generate install manifest - # @ignored - templating: - enabled: false - debug: false - version: ~ - - image: - # -- (string) Global value that allows to set a single image registry across all deployments. - # When set, it will override any values set under `.image.registry` across the chart. - registry: ~ - # -- (list) Global list of Image pull secrets - # When set, it will override any values set under `imagePullSecrets` under different components across the chart. - imagePullSecrets: [] - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - caCertificates: - # -- Global CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - # Individual controller values will override this global value - data: ~ - - # -- Global value to set single volume to be mounted for CA certificates for all deployments. - # Not used when `.Values.global.caCertificates.data` is defined - # Individual controller values will override this global value - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Additional container environment variables to apply to all containers and init containers - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Global node labels for pod assignment. Non-global values will override the global value. - nodeSelector: - cloud.google.com/compute-class: datascience-kyverno - - # -- Global List of node taints to tolerate. Non-global values will override the global value. - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-kyverno" - effect: "NoSchedule" - -# -- (string) Override the name of the chart -nameOverride: ~ - -# -- (string) Override the expanded name of the chart -fullnameOverride: ~ - -# -- (string) Override the namespace the chart deploys to -namespaceOverride: ~ - -upgrade: - # -- Upgrading from v2 to v3 is not allowed by default, set this to true once changes have been reviewed. - fromV2: false - -apiVersionOverride: - # -- (string) Override api version used to create `PodDisruptionBudget`` resources. - # When not specified the chart will check if `policy/v1/PodDisruptionBudget` is available to - # determine the api version automatically. - podDisruptionBudget: ~ - -rbac: - roles: - # -- Aggregate ClusterRoles to Kubernetes default user-facing roles. For more information, see [User-facing roles](https://kubernetes.io/docs/reference/access-authn-authz/rbac/#user-facing-roles) - aggregate: - admin: true - view: true - -# Use openreports.io as the API group for reporting -openreports: - # -- Enable OpenReports feature in controllers - enabled: false - # -- Whether to install CRDs from the upstream OpenReports chart. Setting this to true requires enabled to also be true. - installCrds: false - -# CRDs configuration -crds: - - # -- Whether to have Helm install the Kyverno CRDs, if the CRDs are not installed by Helm, they must be added before policies can be created - install: true - - reportsServer: - # -- Kyverno reports-server is used in your cluster - enabled: false - - groups: - - # -- Install CRDs in group `kyverno.io` - kyverno: - cleanuppolicies: true - clustercleanuppolicies: true - clusterpolicies: true - globalcontextentries: true - policies: true - policyexceptions: true - updaterequests: true - - # -- Install CRDs in group `policies.kyverno.io` - policies: - validatingpolicies: true - policyexceptions: true - imagevalidatingpolicies: true - namespacedimagevalidatingpolicies: true - mutatingpolicies: true - generatingpolicies: true - deletingpolicies: true - namespaceddeletingpolicies: true - namespacedvalidatingpolicies: true - - # -- Install CRDs in group `reports.kyverno.io` - reports: - clusterephemeralreports: true - ephemeralreports: true - - # -- Install CRDs in group `wgpolicyk8s.io` - wgpolicyk8s: - clusterpolicyreports: false ##lk - policyreports: false ##lk - - # -- Additional CRDs annotations (Replace=true avoids 262144-byte annotation limit on large CRDs) - annotations: - argocd.argoproj.io/sync-options: Replace=true - # strategy.spinnaker.io/replace: 'true' - - # -- Additional CRDs labels - customLabels: {} - - migration: - - # -- Enable CRDs migration using helm post upgrade hook - enabled: true - - # -- Resources to migrate - resources: - - cleanuppolicies.kyverno.io - - clustercleanuppolicies.kyverno.io - - clusterpolicies.kyverno.io - - globalcontextentries.kyverno.io - - policies.kyverno.io - - policyexceptions.kyverno.io - - updaterequests.kyverno.io - - deletingpolicies.policies.kyverno.io - - generatingpolicies.policies.kyverno.io - - imagevalidatingpolicies.policies.kyverno.io - - namespacedimagevalidatingpolicies.policies.kyverno.io - - mutatingpolicies.policies.kyverno.io - - namespaceddeletingpolicies.policies.kyverno.io - - namespacedvalidatingpolicies.policies.kyverno.io - - policyexceptions.policies.kyverno.io - - validatingpolicies.policies.kyverno.io - - image: - # -- (string) Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- (string) Image repository - repository: kyverno/kyverno-cli - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- (string) Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podResources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -# Configuration -config: - - # -- Create the configmap. - create: true - - # -- Preserve the configmap settings during upgrade. - preserve: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - # -- Enable registry mutation for container images. Enabled by default. - enableDefaultRegistryMutation: true - - # -- The registry hostname used for the image mutation. - defaultRegistry: docker.io - - # -- Exclude groups - excludeGroups: - - system:nodes - - # -- Exclude usernames - excludeUsernames: [] - # - '!system:kube-scheduler' - - # -- Exclude roles - excludeRoles: [] - - # -- Exclude roles - excludeClusterRoles: [] - - # -- Generate success events. - generateSuccessEvents: false - - # -- Resource types to be skipped by the Kyverno policy engine. - # Make sure to surround each entry in quotes so that it doesn't get parsed as a nested YAML list. - # These are joined together without spaces, run through `tpl`, and the result is set in the config map. - # @default -- See [values.yaml](values.yaml) - resourceFilters: - - '[Event,*,*]' - # - '[*/*,kube-system,*]' - - '[*/*,kube-public,*]' - - '[*/*,kube-node-lease,*]' - - '[Node,*,*]' - - '[Node/?*,*,*]' - - '[APIService,*,*]' - - '[APIService/?*,*,*]' - - '[TokenReview,*,*]' - - '[SubjectAccessReview,*,*]' - - '[SelfSubjectAccessReview,*,*]' - - '[Binding,*,*]' - - '[Pod/binding,*,*]' - - '[ReplicaSet,*,*]' - - '[ReplicaSet/?*,*,*]' - - '[EphemeralReport,*,*]' - - '[ClusterEphemeralReport,*,*]' - # exclude resources from the chart - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:additional]' - - '[ClusterRoleBinding,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.configMapName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.metricsConfigMapName" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Job,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[Job/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.admission-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.background-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - # -- Sets the threshold for the total number of UpdateRequests generated for mutateExisitng and generate policies. - updateRequestThreshold: 1000 - - # -- Defines the `namespaceSelector`/`objectSelector` in the webhook configurations. - # The Kyverno namespace is excluded if `excludeKyvernoNamespace` is `true` (default) - webhooks: - # Exclude namespaces - # namespaceSelector: lk-s2 - # matchExpressions: - # - key: kubernetes.io/metadata.name - # operator: NotIn - # values: - # - kube-systemm lk-s2 - # Exclude objects - # objectSelector: - # matchExpressions: - # - key: webhooks.kyverno.io/exclude - # operator: DoesNotExist - - # -- Defines annotations to set on webhook configurations. - webhookAnnotations: - # Example to disable admission enforcer on AKS: - # 'admissions.enforcer/disabled': 'true' lk-s3 - - # -- Defines labels to set on webhook configurations. - webhookLabels: {} - # Example to adopt webhook resources in ArgoCD: - # 'argocd.argoproj.io/instance': 'kyverno' - - # -- Defines match conditions to set on webhook configurations (requires Kubernetes 1.27+). - matchConditions: [] - - # -- Exclude Kyverno namespace - # Determines if default Kyverno namespace exclusion is enabled for webhooks and resourceFilters - excludeKyvernoNamespace: true - - # -- resourceFilter namespace exclude - # Namespaces to exclude from the default resourceFilters - resourceFiltersExcludeNamespaces: [] - - # -- resourceFilters exclude list - # Items to exclude from config.resourceFilters - resourceFiltersExclude: [] - - # -- resourceFilter namespace include - # Namespaces to include to the default resourceFilters - resourceFiltersIncludeNamespaces: [] - - # -- resourceFilters include list - # Items to include to config.resourceFilters - resourceFiltersInclude: [] - -# Metrics configuration -metricsConfig: - - # -- Create the configmap. - create: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - namespaces: - - # -- List of namespaces to capture metrics for. - include: [] - - # -- list of namespaces to NOT capture metrics for. - exclude: [] - - # -- (string) Rate at which metrics should reset so as to clean up the memory footprint of kyverno metrics, if you might be expecting high memory footprint of Kyverno's metrics. Default: 0, no refresh of metrics. WARNING: This flag is not working since Kyverno 1.8.0 - metricsRefreshInterval: ~ - # metricsRefreshInterval: 24h - - # -- (list) Configures the bucket boundaries for all Histogram metrics, changing this configuration requires restart of the kyverno admission controller - bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 15, 20, 25, 30] - - # -- (map) Configures the exposure of individual metrics, by default all metrics and all labels are exported, changing this configuration requires restart of the kyverno admission controller - metricsExposure: - kyverno_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_image_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_mutating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_generating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_admission_review_duration_seconds: - # enabled: false - disabledLabelDimensions: ["resource_namespace"] - kyverno_policy_rule_info_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_policy_results_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_admission_requests_total: - disabledLabelDimensions: ["resource_namespace"] - kyverno_cleanup_controller_deletedobjects_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - -# -- Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -imagePullSecrets: {} - # regcred: - # registry: foo.example.com - # username: foobar - # password: secret - # regcred2: - # registry: bar.example.com - # username: barbaz - # password: secret2 - -# -- Existing Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -existingImagePullSecrets: [] - # - test-registry - # - other-test-registry - -# Tests configuration -test: - # -- Sleep time before running test - sleep: 20 - - image: - # -- (string) Image registry - registry: curlimages - # -- Image repository - repository: curl - # -- Image tag - # Defaults to `latest` if omitted - tag: '8.10.1' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - # -- Security context for the test containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- Additional Pod annotations - podAnnotations: {} - - # -- List of node taints to tolerate - tolerations: [] - -# -- Additional labels -customLabels: {} - - -webhooksCleanup: - # -- Create a helm pre-delete hook to cleanup webhooks. - enabled: true - - autoDeleteWebhooks: - # -- Allow webhooks controller to delete webhooks using finalizers - enabled: false - - image: - # -- (string) Image registry - registry: registry.k8s.io - # -- Image repository - repository: kubectl - # -- Image tag - # Defaults to `latest` if omitted - tag: 'v1.32.7' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -grafana: - # -- Enable grafana dashboard creation. - enabled: false - - # -- Configmap name template. - configMapName: '{{ include "kyverno.fullname" . }}-grafana' - - # -- (string) Namespace to create the grafana dashboard configmap. - # If not set, it will be created in the same namespace where the chart is deployed. - namespace: ~ - - # -- Grafana dashboard configmap annotations. - annotations: {} - - # -- Grafana dashboard configmap labels - labels: - grafana_dashboard: "1" - - # -- create GrafanaDashboard custom resource referencing to the configMap. - # according to https://grafana-operator.github.io/grafana-operator/docs/examples/dashboard_from_configmap/readme/ - grafanaDashboard: - create: false - folder: kyverno - allowCrossNamespaceImport: true - matchLabels: - dashboards: "grafana" - -# Features configuration -features: - admissionReports: - # -- Enables the feature - enabled: true - aggregateReports: - # -- Enables the feature - enabled: true - policyReports: - # -- Enables the feature - enabled: false ##lk - validatingAdmissionPolicyReports: - # -- Enables the feature - enabled: true - mutatingAdmissionPolicyReports: - # -- Enables the feature - enabled: false - reporting: - # -- Enables the feature - validate: true - # -- Enables the feature - mutate: true - # -- Enables the feature - mutateExisting: true - # -- Enables the feature - imageVerify: true - # -- Enables the feature - generate: true - autoUpdateWebhooks: - # -- Enables the feature - enabled: true - backgroundScan: - # -- Enables the feature - enabled: true - # -- Number of background scan workers - backgroundScanWorkers: 2 - # -- Background scan interval - backgroundScanInterval: 1h - # -- Skips resource filters in background scan - skipResourceFilters: true - configMapCaching: - # -- Enables the feature - enabled: true - controllerRuntimeMetrics: - # -- Bind address for controller-runtime metrics (use "0" to disable it) - bindAddress: ":8080" - deferredLoading: - # -- Enables the feature - enabled: true - dumpPayload: - # -- Enables the feature - enabled: false - forceFailurePolicyIgnore: - # -- Enables the feature - enabled: false - generateValidatingAdmissionPolicy: - # -- Enables the feature - enabled: true - generateMutatingAdmissionPolicy: - # -- Enables the feature - enabled: false - dumpPatches: - # -- Enables the feature - enabled: false - globalContext: - # -- Maximum allowed response size from API Calls. A value of 0 bypasses checks (not recommended) - maxApiCallResponseLength: 2000000 - logging: - # -- Logging format - format: text - # -- Logging verbosity - verbosity: 2 - omitEvents: - # -- Events which should not be emitted (possible values `PolicyViolation`, `PolicyApplied`, `PolicyError`, and `PolicySkipped`) - eventTypes: - - PolicyApplied - - PolicySkipped - # - PolicyViolation - # - PolicyError - policyExceptions: - # -- Enables the feature - enabled: true - # -- Restrict policy exceptions to a single namespace - # Set to "*" to allow exceptions in all namespaces - namespace: '' - protectManagedResources: - # -- Enables the feature - enabled: false - registryClient: - # -- Allow insecure registry - allowInsecure: false - # -- Enable registry client helpers - credentialHelpers: - - default - - google - - amazon - - azure - - github - ttlController: - # -- Reconciliation interval for the label based cleanup manager - reconciliationInterval: 1m - tuf: - # -- Enables the feature - enabled: false - # -- (string) Path to Tuf root - root: ~ - # -- (string) Raw Tuf root - rootRaw: ~ - # -- (string) Tuf mirror - mirror: ~ - -# Admission controller configuration -admissionController: - autoscaling: - # -- Enable horizontal pod autoscaling - enabled: true - - # -- Minimum number of pods - minReplicas: 1 - - # -- Maximum number of pods - maxReplicas: 10 - - # -- Target CPU utilization percentage - targetCPUUtilizationPercentage: 80 - - # -- Configurable scaling behavior - behavior: {} - - # -- Overrides features defined at the root level - featuresOverride: - admissionReports: - # -- Max number of admission reports allowed in flight until the admission controller stops creating new ones - backPressureThreshold: 1000 - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- The ServiceAccount name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno admission controller activities. - # This will help ensure Kyverno stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- admissionController webhook server port - # in case you are using hostNetwork: true, you might want to change the port the webhookServer is listening to - webhookServer: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - admission-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.admissionController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - initContainer: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyvernopre - # -- (string) Image tag - # If missing, defaults to image.tag - tag: ~ - # -- (string) Image pull policy - # If missing, defaults to image.pullPolicy - pullPolicy: ~ - - resources: - # -- Pod resource limits - limits: - cpu: 1000m - memory: 1Gi - # -- Pod resource requests - requests: - cpu: 1000m - memory: 1Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - container: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyverno - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - resources: - # -- Pod resource limits - limits: - cpu: 1900m - memory: 3Gi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Array of extra init containers - extraInitContainers: [] - # - name: init-container - # image: busybox - # command: ['sh', '-c', 'echo Hello'] - - # -- Array of extra containers to run alongside kyverno - extraContainers: [] - # - name: myapp-container - # image: busybox - # command: ['sh', '-c', 'echo Hello && sleep 3600'] - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Kyverno's metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Background controller configuration -backgroundController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable background controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: - - apiGroups: - - networking.k8s.io - resources: - - ingresses - - ingressclasses - - networkpolicies - verbs: - - create - - update - - patch - - delete - - apiGroups: - - rbac.authorization.k8s.io - resources: - - rolebindings - - roles - verbs: - - create - - update - - patch - - delete - - apiGroups: - - '' - resources: - - configmaps - - resourcequotas - - limitranges - verbs: - - create - - update - - patch - - delete - - apiGroups: - - resource.k8s.io - resources: - - resourceclaims - - resourceclaimtemplates - verbs: - - create - - delete - - update - - patch - - deletecollection - - apiGroups: - - autoscaling - resources: - - horizontalpodautoscalers - verbs: - - get - - list - - watch - - update - - patch - - apiGroups: - - apps - resources: - - deployments - - statefulsets - - replicasets - verbs: - - get - - list - - watch - - update - - patch - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - # - patch - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/background-controller - # -- Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 1900m - memory: 3Gi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - background-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.backgroundController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - # -- backgroundController server port - # in case you are using hostNetwork: true, you might want to change the port the backgroundController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Cleanup controller configuration -cleanupController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable cleanup controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - delete - # - list - # - watch - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/cleanup-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- cleanupController server port - # in case you are using hostNetwork: true, you might want to change the port the cleanupController is listening to - server: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 1100m - memory: 1.2Gi - # -- Pod resource requests - requests: - cpu: 1100m - memory: 1.2Gi - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - cleanup-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `service.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Reports controller configuration -reportsController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable reports controller. - enabled: false ##lk - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/reports-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno reports controller activities. - # This will help ensure Kyverno reports stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 200m - memory: 128Mi - # -- Pod resource requests - requests: - cpu: 100m - memory: 64Mi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - reports-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.reportsController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- (string) Service node port. - # Only used if `type` is `NodePort`. - nodePort: ~ - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- (string) Traces receiver address - address: ~ - # -- (string) Traces receiver port - port: ~ - # -- (string) Traces receiver credentials - creds: ~ - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- (string) Otel collector endpoint - collector: ~ - # -- (string) Otel collector credentials - creds: ~ - - # -- reportsController server port - # in case you are using hostNetwork: true, you might want to change the port the reportsController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - - # -- Enable sanity check for reports CRDs - sanityChecks: true diff --git a/helm-overrides/gke-datascience-prd-ase1a/kyverno/policies/protect-namespace.yaml b/helm-overrides/gke-datascience-prd-ase1a/kyverno/policies/protect-namespace.yaml deleted file mode 100644 index 87e2f36..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/kyverno/policies/protect-namespace.yaml +++ /dev/null @@ -1,40 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: protect-critical-namespaces -spec: - validationFailureAction: Enforce - background: false - rules: - - name: deny-deletion-of-critical-namespaces - match: - any: - - resources: - kinds: - - Namespace - names: - - keda-datascience-a-prd - - contour-internal-0-datascience-a-prd - - contour-internal-0-datascience-a-prd-intra - - contour-external-datascience-a-prd - - external-secrets-datascience-a-prd - - flagger-datascience-a-prd - - victoriametrics - - kube-system - - monitoring - - telegraf-operator - - loadtester - - opentelemetry - - fluentd - - kube-events - - observability - - contour-cert-checker-ns - - aurva-dataplane - validate: - message: "Deletion of critical namespaces is not allowed." - deny: - conditions: - any: - - key: "{{request.operation}}" - operator: Equals - value: DELETE diff --git a/helm-overrides/gke-datascience-prd-ase1a/kyverno/policies/restrict-replicas.yaml b/helm-overrides/gke-datascience-prd-ase1a/kyverno/policies/restrict-replicas.yaml deleted file mode 100644 index d456985..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/kyverno/policies/restrict-replicas.yaml +++ /dev/null @@ -1,29 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: restrict-minimum-replicas-selected-ns -spec: - validationFailureAction: Enforce - background: false - rules: - - name: min-replicas-deployments-statefulsets - match: - resources: - kinds: - - Deployment - - StatefulSet - - Deployment/scale - - StatefulSet/scale - names: - - coredns - namespaces: - - kube-system - operations: - - CREATE - - UPDATE - validate: - message: "Deployments and StatefulSets must have at least 5 replicas in this namespace." - anyPattern: - # Case 1: replicas explicitly set and >= 5 - - spec: - replicas: ">=5" diff --git a/helm-overrides/gke-datascience-prd-ase1a/loadtester/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/loadtester/custom-values.yaml deleted file mode 100644 index 37e21dd..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/loadtester/custom-values.yaml +++ /dev/null @@ -1,115 +0,0 @@ -replicaCount: 1 - -Namespace: loadtester - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/dev/devops/flagger-loadtester - tag: 0.30.0 - pullPolicy: IfNotPresent - pullSecret: - -podLabels: - bu: datascience - team: datascience-shared -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - -podPriorityClassName: "" - -logLevel: info -cmd: - timeout: 1h - namespaceRegexp: "" - -nameOverride: "loadtester" -fullnameOverride: "" - -env: [] - -service: - type: ClusterIP - port: 80 - -resources: - requests: - cpu: 10m - memory: 64Mi - -volumes: [] -volumeMounts: [] - -nodeSelector: - cloud.google.com/compute-class: datascience-devops - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - -affinity: {} - -ingress: - slowStart: - enabled: false - enabled: true - ingressClassName: contour-internal-1 - servicePort: http - hosts: - - host: datascience-flagger-loadtester.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / -createContourGateway: true -contourResponseTimeout: false -rbac: - # rbac.create: `true` if rbac resources should be created - create: true - # rbac.scope: `cluster` to create cluster-scope rbac resources (ClusterRole/ClusterRoleBinding) - # otherwise, namespace-scope rbac resources will be created (Role/RoleBinding) - scope: - # rbac.rules: array of rules to apply to the role. example: - # rules: - # - apiGroups: [""] - # resources: ["pods"] - # verbs: ["list", "get"] - rules: [] - -# name of an existing service account to use - if not creating rbac resources -serviceAccountName: "" - -# App Mesh virtual node settings (to be used for AppMesh v1beta1) -meshName: "" -#backends: -# - app1.namespace -# - app2.namespace - -# App Mesh virtual node settings (to be used for AppMesh v1beta2) -appmesh: - enabled: false - backends: - - podinfo - - podinfo-canary - -#Istio virtual service and gatway settings. TLS secrets should be in namespace before enbaled it. ( secret format loadtester.fullname ) -istio: - enabled: false - host: flagger-loadtester.flagger - gateway: - enabled: false - tls: - enabled: false - httpsRedirect: false - -# when enabled, it will add a security context for the loadtester pod -securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 100 - runAsGroup: 101 - -podDisruptionBudget: - enabled: false - minAvailable: 1 diff --git a/helm-overrides/gke-datascience-prd-ase1a/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index 5c3b475..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,153 +0,0 @@ -fullnameOverride: opentelemetry-datascience-a-prd - -mode: daemonset - -priorityClassName: "system-node-critical" - -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-datascience-contour-ext - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-dsci-default-prd-ase1a - - key: cloud.google.com/compute-class - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external - - contour-shared - - contour-intra-0 - - contour-intra-1 - - vm-stack-ht - - vmagent-c4 - - vmagent - - vmagent-mds - - vmagent-n4 - - vminsert - - vminsert-mds - - vmselect - - vmselect-mds - - vmstorage-c4d - - vmstorage - - vmstorage-mds - - vmstorage-n4 - - vmstorage-n4d - - vmstorage-sale-24aug - - vmstorage-tmp - - alloy - - preprod-spot-16 -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 200m - memory: 256Mi - limits: - cpu: 400m - memory: 512Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "datascience" - team: "sre" - service: "opentelemetry-datascience-a-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% diff --git a/helm-overrides/gke-datascience-prd-ase1a/prometheus-node-exporter/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index f2e9646..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,496 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-datascience-a-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "datascience" - team: "datascience-sre" - service: "node-exporter-datascience-a-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/gke-datascience-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index 8146fe1..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-datascience-a-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-datascience-a-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "datascience" - team: "datascience-sre" - service: "stackdriver-exporter-datascience-a-prd" - env: "prd" - priority: "p0" - type: "exporter" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-datascience-prd-0622" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,compute.googleapis.com/instance,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - cloud.google.com/compute-class: "datascience-devops" - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-dssre-stackdriver-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/gke-datascience-prd-ase1a/telegraf-operator/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/telegraf-operator/custom-values.yaml deleted file mode 100644 index 4c7d119..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,232 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - - infra-histogram-optimized: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 80000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - namepass = ["DOWNSTREAM"] - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - namedrop = ["DOWNSTREAM"] - stats = ["count"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - expiration_interval = "1m" - [[aggregators.histogram.config]] - buckets = [10.0, 25.0, 50.0, 100.0, 250.0, 500.0, 1000.0, 2500.0, 5000.0, 10000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "datascience" - team: "datascience-sre" - service: "telegraf-operator-datascience-a-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - cloud.google.com/compute-class: "datascience-devops" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false \ No newline at end of file diff --git a/helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml deleted file mode 100644 index 0380fe5..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml +++ /dev/null @@ -1,295 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-datascience-a-prd-dr -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-datascience-a-prd-dr-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-spsre-vmagent-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - - https://vminsert-infra-prd-dr.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vmagent-datascience-a-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-datascience-a-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-datascience-a-prd-dr.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 12 - memory: 12Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - cloud.google.com/compute-class: "vmagent-dr" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml deleted file mode 100644 index 88e3fbe..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml +++ /dev/null @@ -1,272 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 1 - -fullnameOverride: vmagent-datascience-a-prd-fb -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-datascience-a-prd-fb-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - -# Remove topology spread constraints -topologySpreadConstraints: [] - -# Modify deployment strategy -deployment: - enabled: true - -# Remove HPA -horizontalPodAutoscaler: - enabled: false - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-infr-sre-vmagent-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -# Remove pod disruption budget -podDisruptionBudget: - enabled: false - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-datascience.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vmsinglenode-infra-prd.victoriametrics.svc.clusterset.local:8428/api/v1/write - - https://vmsinglenode-infra-prd.meeshogcp.in/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "datascience-fb" - team: "sre" - service: "vmagent-datascience-a-prd-fb" - env: "prd" - priority: "p0" - type: "vmagent-fb" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-datascience-a-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-datascience-a-prd-fb.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -# Modify resources -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - requests: - cpu: 2 - memory: 3Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - cloud.google.com/compute-class: "vmagent-n4" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-n4" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml deleted file mode 100644 index cbb07fb..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-datascience-a-prd-dr - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/datascience/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-datascience-a-prd-dr.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-datascience-a-prd-dr" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmagent-dr" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-alert-secured-stateful-v0/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-alert-secured-stateful-v0/custom-values.yaml deleted file mode 100644 index f70d7ac..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-alert-secured-stateful-v0/custom-values.yaml +++ /dev/null @@ -1,332 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-stateful-secured-datascience-a-prd-v0 - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-datascience-a-prd-proxy-v0.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-datascience-a-prd-v0.victoriametrics.svc.cluster.local:8480/insert/200/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-datascience-a-prd-proxy-v0.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/sensitive/datascience/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: [] - # Extra Volume Mounts for the container - extraVolumeMounts: [] - extraContainers: [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: false - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-datascience-a-prd-stateful-v0.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "datascience" - team: "sre" - service: "vmalert-stateful-secured-datascience-a-prd-v0" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: "false" - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmselect" - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-secured-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-alert-stateful-v0/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-alert-stateful-v0/custom-values.yaml deleted file mode 100644 index c34c7a7..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/victoria-metrics-alert-stateful-v0/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-datascience-a-prd-stateful-v0 - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-datascience-a-prd-proxy-v0.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-datascience-a-prd-v0.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-datascience-a-prd-proxy-v0.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/datascience/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-datascience-a-prd-stateful-v0.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 1500m - memory: 3Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "datascience" - team: "sre" - service: "vmalert-datascience-a-prd-stateful-v0" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmselect" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-datascience-prd-ase1a/victoriametrics-agent-v0/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/victoriametrics-agent-v0/custom-values.yaml deleted file mode 100644 index 196c54c..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/victoriametrics-agent-v0/custom-values.yaml +++ /dev/null @@ -1,488 +0,0 @@ -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "datascience-a-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -# -- Replica count -replicaCount: 2 - -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - -# -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables -deployment: - spec: - minReadySeconds: 1200 - progressDeadlineSeconds: 1800 - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - -image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - # -- Image tag, set to `Chart.AppVersion` by default - tag: v1.133.0 # rewrites Chart.AppVersion - # -- Variant of the image to use. - # e.g. enterprise, scratch - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - -# -- Image pull secrets -imagePullSecrets: [] - -# -- Add additional DNS entries to pods hosts file. Check [official documentation](https://kubernetes.io/docs/tasks/network/customize-hosts-file-for-pods/) -hostAliases: [] -# - ip: 192.168.1.1 -# hostNames: -# - test.example.com -# - another.example.net - -# -- Override chart name -nameOverride: "" - -# -- Override resources fullname -fullnameOverride: "vm-agent-datascience-a-prd-v0" - -# -- Container working directory -containerWorkingDir: "/" - -rbac: - # -- Enables Role/RoleBinding creation - create: true - - # -- Role/RoleBinding annotations - annotations: {} - - # -- Role/RoleBinding labels - extraLabels: {} - - # -- If true and `rbac.enabled`, will deploy a Role/RoleBinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - - # -- additional rules for a role - extraRules: [] - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - # -- Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-sre-vmagnt-prd-mds@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # -- mount API token to pod directly - automountToken: true - -# -- See `kubectl explain poddisruptionbudget.spec` for more or check [official documentation](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# -- Generates `remoteWrite.*` flags and config maps with value content for values, that are of type list of map. -# Each item should contain `url` param to pass validation. -remoteWrite: - # - url: http://vm-insert-datascience-a-prd-v0.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write - # disableOnDiskQueue: false - # dropSamplesOnOverload: false - - url: http://vm-insert-datascience-prd.victoriametrics.svc.clusterset.local:8480/insert/multitenant/prometheus/api/v1/write - disableOnDiskQueue: false - dropSamplesOnOverload: false - - url: http://prd-census-server-datascience.prd-census-server-datascience.svc.cluster.local/api/v1/write - disableOnDiskQueue: true - dropSamplesOnOverload: true -# urlRelabelConfig: -# - action: keep -# source_labels: [env] -# regex: "dev" - -# -- VMAgent extra command line arguments -extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8429 - promscrape.config.strictParse: false # default is true - promscrape.maxScrapeSize: 1000000000 # default is 167772160 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - # promscrape.suppressScrapeErrors: true # default is false - - # Uncomment and specify the port if you want to support any of the protocols: - # https://docs.victoriametrics.com/victoriametrics/vmagent/#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for more details. -env: - - name: GOGC - value: "200" - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth-secret - # key: password - -# -- Specify alternative source for env variables -envFrom: - [] - #- configMapRef: - # name: special-config - -# -- Extra labels for Deployment and Statefulset -extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-agent-datascience-a-prd-v0" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra labels for Pods only -podLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-agent-datascience-a-prd-v0" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra selector labels common for pod and service -selectorLabels: {} - -# -- Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# -- Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# -- Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -# -- Extra containers to run in a pod with vmagent -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -# -- Init containers for vmagent -initContainers: - [] - # - name: example - # image: example-image - -# -- Security context to be added to pod -podSecurityContext: - enabled: true - # fsGroup: 2000 - -# -- Security context to be added to pod's containers -securityContext: - enabled: true - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - # -- Enable agent service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - extraLabels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) for details - externalIPs: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8429 - # -- Target port - targetPort: http - # nodePort: 30000 - # -- Service type - type: ClusterIP - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - -ingress: - # -- Enable deployment of ingress for agent - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-agent-datascience-a-prd-v0.meeshogcp.in - path: - - / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - - # -- Ingress controller class name - ingressClassName: "nginx-internal" - - # -- Ingress path type - pathType: Prefix - -# -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) -resources: - requests: - cpu: 70 - memory: 100Gi - -# -- Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) -nodeSelector: - cloud.google.com/compute-class: "vmagent-n4" - -# -- Node tolerations for server scheduling to nodes with taints. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-n4" - effect: "NoSchedule" - -# -- Pod topologySpreadConstraints -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# -- Pod affinity -affinity: {} - -# -- VMAgent [scraping configuration](https://docs.victoriametrics.com/victoriametrics/vmagent/#how-to-collect-metrics-in-prometheus-format) -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-datascience-a-prd-v0" - -# -- Priority class to be assigned to the pod(s) -priorityClassName: "" - -# -- Enable the host network -hostNetwork: false - -serviceMonitor: - # -- Enable deployment of Service Monitor for server component. This is Prometheus operator object - enabled: false - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Service Monitor relabelings - relabelings: [] - # -- Basic auth params for Service Monitor - basicAuth: {} - # -- Service Monitor metricRelabelings - metricRelabelings: [] - # -- Service Monitor targetPort - targetPort: http - # interval: 15s - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - -# -- Empty dir configuration for a case, when persistence is disabled -emptyDir: {} - -persistentVolume: - # -- Create/use Persistent Volume Claim for server component. Empty dir if false - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Size of the volume. Should be calculated based on the logs you send and retention policy you set. - size: 10Gi - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume additional labels - extraLabels: {} - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Horizontal Pod Autoscaling. -# Note that it is not intended to be used for vmagents which perform scraping. -# In order to scale scraping vmagents check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) -horizontalPodAutoscaling: - # -- Use HPA for vmagent - enabled: false - # -- Maximum replicas for HPA to use to to scale vmagent - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale vmagent - minReplicas: 1 - # -- Metric for HPA to use to scale vmagent - metrics: [] - -# -- VMAgent scrape configuration -config: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -probe: - # -- Readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - # -- Liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - # -- Startup probe - startup: {} - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -allowedMetricsEndpoints: - - /metrics - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-datascience-prd-ase1a/victoriametrics-insert-v0/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/victoriametrics-insert-v0/custom-values.yaml deleted file mode 100644 index f9d8de5..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/victoriametrics-insert-v0/custom-values.yaml +++ /dev/null @@ -1,411 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: false - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- IDs of vmstorage nodes to exclude from writing - excludeStorageIDs: [] - # -- Override default `app` label name - name: vminsert - # -- VMInsert strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - # override Chart.AppVersion - tag: v1.107.0-cluster - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMInsert http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-datascience-a-prd-v0 - # -- Extra command line arguments for vminsert component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8480 - maxLabelsPerTimeseries: 65 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-datascience-a-prd-v0-0.vm-storage-datascience-a-prd-v0.victoriametrics.svc:8400" - - "vm-storage-datascience-a-prd-v0-1.vm-storage-datascience-a-prd-v0.victoriametrics.svc:8400" - - "vm-storage-datascience-a-prd-v0-2.vm-storage-datascience-a-prd-v0.victoriametrics.svc:8400" - - "vm-storage-datascience-a-prd-v0-3.vm-storage-datascience-a-prd-v0.victoriametrics.svc:8400" - - "vm-storage-datascience-a-prd-v0-4.vm-storage-datascience-a-prd-v0.victoriametrics.svc:8400" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-insert-datascience-a-prd-v0" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Pod's additional labels - podLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-insert-datascience-a-prd-v0" - env: "prd" - priority: "p0" - type: "vminsert" - terminationGracePeriodSeconds: 30 - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - - # -- Readiness & Liveness probes - probe: - # -- VMInsert readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert startup probe - startup: {} - - # -- Relabel configuration - relabel: - enabled: false - config: [] - # -- Use existing configmap if specified - # otherwise .config values will be used. Relabel config **should** reside under `relabel.yml` key - configMap: "" - - # Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 8 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vminsert - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vminsert - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - cloud.google.com/compute-class: "vminsert-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 8 - memory: 8Gi - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - service: - # -- Create VMInsert service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here]( https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) for details - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-insert-datascience-a-prd-v0.meeshogcp.in - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - - # -- Ingress controller class name - ingressClassName: nginx-internal - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for insert component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/insert" .Values.vminsert }}' - - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vminsert component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-datascience-prd-ase1a/victoriametrics-select-v0/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/victoriametrics-select-v0/custom-values.yaml deleted file mode 100644 index 74c6ebf..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/victoriametrics-select-v0/custom-values.yaml +++ /dev/null @@ -1,449 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - - splitService: true - externalService: - enabled: true - name: vmselect-datascience-a-prd-proxy - - # -- Enable clusternative service for LoadBalancer vmselect component - clusternativeService: - enabled: false - - # -- Override default `app` label name - name: "" - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # override Chart.AppVersion - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMSelect http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-datascience-a-prd-v0 - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - # -- Extra command line arguments for vmselect component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8481 - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-datascience-a-prd-v0-0.vm-storage-datascience-a-prd-v0.victoriametrics.svc:8401" - - "vm-storage-datascience-a-prd-v0-1.vm-storage-datascience-a-prd-v0.victoriametrics.svc:8401" - - "vm-storage-datascience-a-prd-v0-2.vm-storage-datascience-a-prd-v0.victoriametrics.svc:8401" - - "vm-storage-datascience-a-prd-v0-3.vm-storage-datascience-a-prd-v0.victoriametrics.svc:8401" - - "vm-storage-datascience-a-prd-v0-4.vm-storage-datascience-a-prd-v0.victoriametrics.svc:8401" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-select-datascience-a-prd-v0" - env: "prd" - priority: "p0" - type: "vmselect" - # -- Pod's additional labels - podLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-select-datascience-a-prd-v0" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Readiness & Liveness probes - probe: - # -- VMSelect readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 40 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 15 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vmselect - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vmselect - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - cloud.google.com/compute-class: "vmselect" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 18 - memory: 32Gi - - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Create VMSelect service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: false - - # -- Ingress annotations - annotations: {} - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-select-datascience.prd.meesho.int - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # -- Ingress controller class name - ingressClassName: contour-internal-0 - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for select component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/select" .Values.vmselect }}' - - # -- vmselect mode: deployment, daemonSet - mode: deployment - - # -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables - deployment: - spec: - # -- VMSelect strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - - # -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables - statefulSet: - enabled: false - spec: - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume extra labels - extraLabels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmselect component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: false - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - diff --git a/helm-overrides/gke-datascience-prd-ase1a/victoriametrics-storage-v0/custom-values.yaml b/helm-overrides/gke-datascience-prd-ase1a/victoriametrics-storage-v0/custom-values.yaml deleted file mode 100644 index f6d908f..0000000 --- a/helm-overrides/gke-datascience-prd-ase1a/victoriametrics-storage-v0/custom-values.yaml +++ /dev/null @@ -1,363 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - enabled: false - -vminsert: - enabled: false - -vmauth: - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- Override default `app` label name - name: vmstorage - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Variant of the image to use. e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMStorage http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-datascience-a-prd-v0 - - # -- Additional environment variables (ex.: secret tokens, flags). Check https://docs.victoriametrics.com/victoriametrics/#environment-variables for details - env: [] - # -- Specify alternative source for env variables - envFrom: [] - # -- Data retention period. Possible units character: h(ours), d(ays), w(eeks), y(ears), if no unit character specified - month. The minimum retention period is 24h. - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - httpListenAddr: :8482 - loggerTimezone: "Asia/Kolkata" - dedup.minScrapeInterval: 60s - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vvmstorage-n4" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - cloud.google.com/compute-class: "vmstorage-n4" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - # -- Use an alternate scheduler, e.g. "stork". Check https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ for details - # schedulerName: - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Override Persistent Volume Claim name - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are https://kubernetes.io/docs/concepts/storage/persistent-volumes/ - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume extra labels - extraLabels: {} - # -- Storage class name. Will be empty if not set - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume - size: 7400Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-storage-datascience-a-prd-v0" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - # -- Pod's additional labels - podLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-storage-datascience-a-prd-v0" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Details are https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 58 - memory: 475Gi - - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: - enabled: false - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - podSecurityContext: - enabled: false - service: - enabled: true - # -- Service annotations - annotations: {} - # -- Service ClusterIP - clusterIP: None - # -- Service type - type: ClusterIP - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmselect - vmselectPort: 8401 - # -- Extra service ports - extraPorts: [] - # -- Health check node port for a service. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilyPolicy: "" - # -- List of service IP families. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilies: [] - # -- Traffic Distribution. Check https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution - trafficDistribution: "" - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - minReadySeconds: 5 - # -- Readiness probes - probe: - # -- VMStorage readiness probe - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMStorage startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmstorage component - enabled: false - # -- Maximum replicas for HPA to use to to scale the vmstorage component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmstorage component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmstorage component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: - scaleDown: - selectPolicy: Disabled - - vmbackupmanager: - # -- Enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-demand-prd-ase1a/README.md b/helm-overrides/gke-demand-prd-ase1a/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/gke-demand-prd-ase1a/alloy/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/alloy/custom-values.yaml deleted file mode 100644 index 43c8161..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/alloy/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -fullnameOverride: "alloy-demand-prd" - -alloy: - configMap: - configFile: demand.alloy - - clustering: - enabled: true - - extraPorts: - - name: "otlp-grpc" - port: 4317 - targetPort: 4317 - protocol: "TCP" - - name: "otlp-http" - port: 4318 - targetPort: 4318 - protocol: "TCP" - - resources: - requests: - cpu: 14 - memory: 115Gi - -configReloader: - enabled: true - -serviceAccount: - annotations: { - iam.gke.io/gcp-service-account: sa-dmnd-sre-grafna-obs-stk-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - } - -controller: - # Must be one of 'daemonset', 'deployment', or 'statefulset'. - type: 'deployment' - - nodeSelector: - cloud.google.com/compute-class: "alloy" - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "alloy" - effect: "NoSchedule" - - autoscaling: - enabled: true - minReplicas: 1 - maxReplicas: 50 - targetCPUUtilizationPercentage: 80 - targetMemoryUtilizationPercentage: 80 \ No newline at end of file diff --git a/helm-overrides/gke-demand-prd-ase1a/aurva-dataplane/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/aurva-dataplane/custom-values.yaml deleted file mode 100644 index f67e97a..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/aurva-dataplane/custom-values.yaml +++ /dev/null @@ -1,686 +0,0 @@ -## Aurva Data Plane -## Ref: https://github.com/aurva-io/aurva-charts.git - -postgresql: - enabled: true - fullnameOverride: "aurva-dataplane-database" - volumePermissions: - ## @param volumePermissions.enabled Enable init container that changes the owner and group of the persistent volume - ## - enabled: true - global: - storageClass: hyperdisk-balanced - postgresql: - auth: - postgresPassword: "aurva" - database: "controller" - # Add toleration to make sure where this postgres db pod should reside (Applicable for production workloads): For more detail ref: https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/ - primary: - extendedConfiguration: | - max_connections = 300 - - #PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: demand-devops - - # -- Select nodes to deploy which matches the following labels - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: demand-devops - -# -- Provide a name in place of `aurva` -# namespaceOverride: aurva-dataplane - -########################################################## -# Global Configs -########################################################## -global: - aurva_controller: - enabled: true - aurva_fastdet: - enabled: false - aurva_pii_analyzer: - enabled: true - aurva_ocr: - enabled: true - aurva_collector: - enabled: true - - deploymentAnnotations: {} - - priorityClassName: "" - -########################################################## -# Aurva Controller -########################################################## -aurva_controller: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "demand" - team: "demand-devops" - service: "aurva-demand-a-prd" - env: "prd" - priority: "p0" - type: "aurva_controller" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 10 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - image: - # Image of the app container - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-controller/aurva-controller - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - - type: secret - name: aurva-controller-secrets - # - type: configmap - # name: proxy-datasource-config - - # -- Resources to be defined for pod - resources: - limits: - memory: 2.5Gi - cpu: 2 - requests: - memory: 2Gi - cpu: 1 - - aurvaFastdet: - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-fastdet/aurva-fastdet - tag: "v2.30.13" - pullPolicy: IfNotPresent - envFrom: [] - env: [] - resources: - limits: - cpu: 0.5 - memory: 512Mi - requests: - cpu: 0.5 - memory: 512Mi - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: demand-devops - - # -- Taint tolerations for nodes - ##PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: demand-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-controller-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - config: - #variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - COMMAND_URL: "command.aurva-prd.meeshogcp.in:80" - DEPLOYMENT_TYPE: "kubernetes" - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - FLUSHER_WORKER_POOL_SIZE: "1000" - FLUSHER_BATCH_SIZE: "30000" - UNIQUENESS_IDENTIFIER: "gke-demand-prd-ase1a" #Recommendation: should be equal to cluster name - PROVIDER_ACCOUNT_ID: "meesho-demand-prd-0622" # GCP PROJECT ID (not Number) - REGION: "asia-southeast1" #eg: asia-south1 - ENVIRONMENT: "prod" - OCR_ENABLED: "true" - MONITORING_ENABLED: "false" - HYBRID_ONLY_MODE: "false" - FORCE_TLS: "false" - FASTDET_FLAG : "true" - AADHAAR_ENHANCER: "0" - WORKSPACE_EVENT_TRACKING_ENABLED: "false" - ACCESS_IQ_ENABLED: "true" - GRPC_ENFORCE_ALPN_ENABLED: "false" - ENABLE_FASTDET: "true" - FASTDET_MAX_BATCH_SIZE: "100" - HEARTBEAT_INTERVAL: "5m" - #pii data - PII_BUCKET_NAME: "gcs-infra-devop-aurva-demand-prd" - PII_LOG_BUCKET_REGION: "asia-southeast1" - PII_LOG_CRON: "*/5 * * * *" - ENABLE_PII_LOG: "true" - PII_EVIDENCE_UPLOAD_MAX_BLOCKING_TASKS: "50000" - PII_EVIDENCE_UPLOAD_MAX_CONCURRENT_TASKS: "500" - PII_EVIDENCE_MAX_CACHE_WEIGHT: "100" - QUOTA_CLEANUP_CRON: "0 0 * * *" - ENABLE_PII_QUOTA: "true" - MAX_PII_EVIDENCES_PER_KEY_PER_WINDOW: "3" - PII_QUOTA_SYNC_CRON: "*/2 * * * *" - QUOTA_WINDOW_HOURS: "24" - #constants - SKIP_NAMESPACES: "contour-internal-1-demand-a-prd,contour-internal-0-demand-a-prd-intra,contour-internal-0-demand-a-prd,contour-external-demand-a-prd" - CLOUD_PROVIDER: "gcp" - LOG_ENV: "production" - RDS_SCANNER_AVAILABILITY : "false" - REDSHIFT_SCANNER_AVAILABILITY : "false" - S3_SCANNER_AVAILABILITY : "false" - DYNAMO_SCANNER_AVAILABILITY: "false" - DOCDB_SCANNER_AVAILABILITY: "false" - OPENSEARCH_SCANNER_AVAILABILITY: "false" - CLOUDSQL_SCANNER_AVAILABILITY: "true" - BIGQUERY_SCANNER_AVAILABILITY: "true" - AWS_SNAPSHOT_SCANNER_AVAILABILITY: "false" - CLOUDSTORAGE_SCANNER_AVAILABILITY: "true" - KEYSPACES_SCANNER_AVAILABILITY: "false" - ALLOYDB_SCANNER_AVAILABILITY: "true" - BIGTABLE_SCANNER_AVAILABILITY: "true" - GCP_BACKUP_AVAILABILITY: "true" - EGRESS_MODE_ONLY: "false" - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - SCAN_UUID_ENABLED: "true" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-controller-sa - # -- Annotations applied to created service account - annotations: - iam.gke.io/gcp-service-account: sa-demand-prd-aurva-controller@meesho-demand-prd-0622.iam.gserviceaccount.com -# eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 15 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 - - -########################################################## -# Aurva OCR -########################################################## -aurva_ocr: - # -- Additional labels for aurva-controller - additionalLabels: - bu: "demand" - team: "demand-devops" - service: "aurva-demand-a-prd" - env: "prd" - priority: "p0" - type: "aurva_ocr" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 1 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-ocr/aurva-ocr - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-ocr: - type: secret - name: aurva-ocr-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 1 - requests: - memory: 2Gi - cpu: 1 - - # -- Select nodes to deploy which matches the following labels - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: demand-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: demand-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-ocr-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - OCR_TIME_LIMIT: "1" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "gke-demand-prd-ase1a" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-ocr-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # iam.gke.io/gcp-service-account: service-account@gcp.iam.gserviceaccount.com - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva Collector -########################################################## -aurva_collector: - - # -- Additional labels for aurva-analyzer - additionalLabels: - bu: "demand" - team: "demand-devops" - service: "aurva-demand-a-prd" - env: "prd" - priority: "p0" - type: "aurva_collector" - - - # -- Annotations on aurva-analyzer - annotations: {} - # "key": "value" - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-collector/aurva-collector - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-controller: - type: secret - name: aurva-collector-secrets - - resources: - limits: - cpu: 800m - memory: 800Mi - requests: - cpu: 200m - memory: 512Mi - - podSecurityContext: {} - - securityContext: - privileged: true - capabilities: - add: - # For kernel v5.8 and above we don't need CAP_SYS_ADMIN or CAP_SYS_RESOURCE - # we just need CAP_BPF and CAP_PERFMON. This has been tested on our EKS node - # which is on kernel v5.10.x - # When SSL Tracing is required we need CAP_SYS_ADMIN and CAP_SYS_PTRACE - # on top of the previous capabilities - # So finally these are the 4 possible combinations for capabilies - # 1. Newer Kernels without SSL - # - BPF - # - PERFMON - # 2. Newer Kernels with SSL - - SYS_ADMIN - - SYS_PTRACE - # 3. Older Kernels without SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # 4. Older Kernels with SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # - SYS_PTRACE - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - volumes: - - name: debugfs - mountPath: /sys/kernel/debug - hostPath: /sys/kernel/debug - - name: vmlinux - mountPath: /sys/kernel/btf/vmlinux - hostPath: /sys/kernel/btf/vmlinux - - name: procfs - mountPath: /host/proc - hostPath: /proc - - name: bpffs - mountPath: /sys/fs/bpf - hostPath: /sys/fs/bpf - - # -- Taint tolerations for nodes - tolerations: - # - effect: NoSchedule - # key: dedicated - # operator: Equal - # value: megatetra - - operator: Exists - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vm-agent - - vmagent-c4d - - vminsert-mds - - vmselect-mds - - vmstorage-n4d - - contour-external-cc-v1 - - contour-external - - contour-internal-0-cc-v1 - - contour-internal-0 - - contour-internal-1-cc-v1 - - contour-internal-1 - - contour-intra-0-cc-v1 - - contour-intra-0 - - contour-intra-1-cc-v1 - - contour-intra-1 - - contour-shared-cc-v1 - - contour-shared - - alloy - - preprod-spot-16 - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-collector-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - # variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "gke-demand-prd-ase1a" - DEPLOYMENT_TYPE: "kubernetes" - TRACE_INTERNAL_SVC: "true" - TRACE_INTERNAL_SVC_HTTP: "true" - INTERNAL_SVC_SAMPLE_INTERVAL: "10m" - LOGS_TTL: "1h" - TRACE_HTTP2: "true" - TRACE_SSL: "false" - TRACE_PSQL: "false" - TRACE_SQLSERVER: "false" - TRACE_MYSQL: "false" - TRACE_EGRESS: "true" - TRACE_GO_TLS: "false" - TRACE_ML_SERVICES: "false" - # constants - LOG_ENV: production - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - MONITORING_ENABLED: "false" - ENABLE_INGRESS_INFORMER: "false" - ENABLE_SERVICE_INFORMER: "false" - ENABLE_ISTIO_INFORMER: "false" - EXCLUDED_PII_REGEX_TYPES: "ip_address,us_bank_number,us_driver_license,us_itin,us_passport,us_routing,us_mbi,ssn" - AGGREGATOR_MAX_CONNECTIONS: "1000" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-collector-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva PII Analyzer -########################################################## -aurva_pii_analyzer: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "demand" - team: "demand-devops" - service: "aurva-demand-a-prd" - env: "prd" - priority: "p0" - type: "aurva_pii_analyzer" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 3 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - ##PLACEHOLDER## - nodeSelector: - cloud.google.com/compute-class: demand-devops - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-piianalyzer/aurva-piianalyzer - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-pii-analyzer: - type: secret - name: aurva-pii-analyzer-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 4Gi - cpu: 4 - requests: - memory: 2Gi - cpu: 2 - - # nodeSelector: ##PLACEHOLDER## - # cloud.google.com/compute-class: demand-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: demand-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-pii-analyzer-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - SCHEDULER_TIME: "1" - SUPPORTED_REGION: "US" - MAX_HOURS_TO_KEEP_REGARDLESS_OF_ACK_STATUS : "2880" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "gke-demand-prd-ase1a" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-pii-analyzer-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 3 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 diff --git a/helm-overrides/gke-demand-prd-ase1a/cert-manager/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/cert-manager/custom-values.yaml deleted file mode 100644 index 12e9cc3..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/cert-manager/custom-values.yaml +++ /dev/null @@ -1,129 +0,0 @@ -global: - logLevel: 2 - rbac: - create: true - priorityClassName: "high-priority" -installCRDs: false - -crds: - enabled: true - keep: true - -# Cert-manager Controller -replicaCount: 1 -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-controller - tag: v1.20.2 - pullPolicy: IfNotPresent - -nodeSelector: - cloud.google.com/compute-class: demand-devops - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "demand-devops" - effect: "NoSchedule" - -resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 500m - memory: 512Mi - -# Webhook Configuration -webhook: - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-webhook - tag: v1.20.2 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: demand-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "demand-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# CA Injector Configuration -cainjector: - enabled: true - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-cainjector - tag: v1.20.2 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: demand-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "demand-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# ACME Solver Configuration -acmesolver: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-acmesolver - tag: v1.20.2 - pullPolicy: IfNotPresent - -# Startup API Check -startupapicheck: - enabled: true - timeout: 1m - backoffLimit: 4 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-startupapicheck - tag: v1.20.2 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: demand-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "demand-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - -# Prometheus Monitoring -prometheus: - enabled: true - servicemonitor: - enabled: false - interval: 60s - scrapeTimeout: 30s - labels: - prometheus: cert-manager diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/azul-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/azul-cc.yaml deleted file mode 100644 index 9eb4ba0..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/azul-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: azul -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: azul - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c2-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/compactduo-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/compactduo-cc.yaml deleted file mode 100644 index ba310ac..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/compactduo-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: compactduo -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: compactduo - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c4-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: c4-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: c4-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/compacttetra-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/compacttetra-cc.yaml deleted file mode 100644 index fda1a52..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/compacttetra-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: compacttetra -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: compacttetra - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4-standard-8 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4-standard-8 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4-standard-8 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-external-cc-v1.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-external-cc-v1.yaml deleted file mode 100644 index e7df224..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-external-cc-v1.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-external-cc-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-external-cc-v1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-4 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-4 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-external-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-external-cc.yaml deleted file mode 100644 index 5a7a898..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-external-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-external -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-external - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4d-highcpu-4 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-internal-0-cc-v1.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-internal-0-cc-v1.yaml deleted file mode 100644 index e05f0cc..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-internal-0-cc-v1.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-0-cc-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-internal-0-cc-v1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-internal-0-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-internal-0-cc.yaml deleted file mode 100644 index 6e4abba..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-internal-0-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-0 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-internal-0 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-internal-1-cc-v1.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-internal-1-cc-v1.yaml deleted file mode 100644 index 9e7c2ba..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-internal-1-cc-v1.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-1-cc-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-internal-1-cc-v1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-internal-1-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-internal-1-cc.yaml deleted file mode 100644 index c820f5c..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-internal-1-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-internal-1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-intra-0-cc-v1.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-intra-0-cc-v1.yaml deleted file mode 100644 index 66081ae..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-intra-0-cc-v1.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-0-cc-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-intra-0-cc-v1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-intra-0-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-intra-0-cc.yaml deleted file mode 100644 index 9e0cac4..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-intra-0-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-0 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-intra-0 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-intra-1-cc-v1.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-intra-1-cc-v1.yaml deleted file mode 100644 index 916ed1d..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-intra-1-cc-v1.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-1-cc-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-intra-1-cc-v1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-intra-1-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-intra-1-cc.yaml deleted file mode 100644 index a785f03..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-intra-1-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-intra-1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-shared-cc-v1.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-shared-cc-v1.yaml deleted file mode 100644 index cf85586..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-shared-cc-v1.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-shared-cc-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-shared-cc-v1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-shared-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-shared-cc.yaml deleted file mode 100644 index 35427d1..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/contour-shared-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-shared -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-shared - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/demand-devops-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/demand-devops-cc.yaml deleted file mode 100644 index b961cd0..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/demand-devops-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: demand-devops -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: demand-devops - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: e2-standard-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - - location: - zones: - - asia-southeast1-c - machineType: e2-standard-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - - location: - zones: - - asia-southeast1-b - machineType: e2-standard-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/demand-kyverno-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/demand-kyverno-cc.yaml deleted file mode 100644 index 0152054..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/demand-kyverno-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: demand-kyverno -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: demand-kyverno - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - - location: - zones: - - asia-southeast1-c - machineType: n2-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - - location: - zones: - - asia-southeast1-b - machineType: n2-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/dmnd-c4a-32c64g-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/dmnd-c4a-32c64g-cc.yaml deleted file mode 100644 index b822f3e..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/dmnd-c4a-32c64g-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dmnd-c4a-32c64g -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: dmnd-c4a-32c64g - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c4a-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 256 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: c4a-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 256 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: c4a-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 256 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/dns-coldstart-probe-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/dns-coldstart-probe-cc.yaml deleted file mode 100644 index dcaa484..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/dns-coldstart-probe-cc.yaml +++ /dev/null @@ -1,51 +0,0 @@ -# TEMPORARY - diagnostic only. Delete after the investigation closes. -# -# Purpose: reproduce the NodeLocal DNSCache cold-start race behind GCP support -# case 73226135. On a freshly provisioned node, the node-cache container is not -# serving DNS for 22-67s (p50 29s), while application containers start at -# ~t+13s and issue a single un-retried lookup against the kube-dns ClusterIP. -# That lookup times out -> 646 DNS failures/24h across demand+supply prd. -# -# Isolation - NAP automatically taints every node in the auto-created pool: -# cloud.google.com/compute-class=dns-coldstart-probe:NoSchedule -# The taint value is the class name, so it is unique by construction. A pod -# reaches this node only if it carries both that exact nodeSelector and the -# matching toleration, and no other workload in the cluster names this class. -# No extra taint is declared here: the only thing that gets past the -# compute-class taint is a blanket "operator: Exists" toleration, which would -# equally tolerate any custom taint we added. -# DaemonSets still schedule here via their blanket Exists toleration. That is -# intentional and required - node-local-dns is the component under test. -# -# Cost: creates ZERO nodes until a pod selects this class. One e2-standard-4 in -# a single zone while the probe runs; scales back to zero when it is removed. -# maxPodsPerNode is 32 to match prd, so DaemonSet pod-slot pressure is faithful. -# -# Owner: aryaman.parida -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dns-coldstart-probe -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: dns-coldstart-probe - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: e2-standard-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/megaduo-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/megaduo-cc.yaml deleted file mode 100644 index a21da60..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/megaduo-cc.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaduo -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megaduo - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3-highcpu-22 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-a - machineType: c4-highcpu-24 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: c3-highcpu-22 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: c4-highcpu-24 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: c3-highcpu-22 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: c4-highcpu-24 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/megaduo-spp-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/megaduo-spp-cc.yaml deleted file mode 100644 index 84b7175..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/megaduo-spp-cc.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaduo-spp -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megaduo-spp - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3-highcpu-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - - location: - zones: - - asia-southeast1-a - machineType: c4-highcpu-24 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: c3-highcpu-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - - location: - zones: - - asia-southeast1-c - machineType: c4-highcpu-24 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: c3-highcpu-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - - location: - zones: - - asia-southeast1-b - machineType: c4-highcpu-24 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/megaduolite-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/megaduolite-cc.yaml deleted file mode 100644 index 68e1114..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/megaduolite-cc.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaduolite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megaduolite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-custom-16-32768 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-a - machineType: n4-highcpu-16 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n2-custom-16-32768 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: n4-highcpu-16 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n2-custom-16-32768 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: n4-highcpu-16 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/megatetra-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/megatetra-cc.yaml deleted file mode 100644 index f507774..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/megatetra-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megatetra -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megatetra - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3-standard-22 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: c3-standard-22 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: c3-standard-22 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/megatetra-trnst-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/megatetra-trnst-cc.yaml deleted file mode 100644 index d639c7e..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/megatetra-trnst-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megatetra-trnst -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megatetra-trnst - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3-standard-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - - location: - zones: - - asia-southeast1-c - machineType: c3-standard-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - - location: - zones: - - asia-southeast1-b - machineType: c3-standard-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/megatetralite-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/megatetralite-cc.yaml deleted file mode 100644 index 5ab7eb6..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/megatetralite-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megatetralite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megatetralite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/pdp-relay-v1-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/pdp-relay-v1-cc.yaml deleted file mode 100644 index 9caebe5..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/pdp-relay-v1-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: pdp-relay-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: pdp-relay-v1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-highcpu-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/prod-comms-consumer-notification-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/prod-comms-consumer-notification-cc.yaml deleted file mode 100644 index ab88ffe..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/prod-comms-consumer-notification-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: prod-comms-consumer-notification -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: prod-comms-consumer-notification - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/prod-comms-consumer-v1-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/prod-comms-consumer-v1-cc.yaml deleted file mode 100644 index 26f199c..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/prod-comms-consumer-v1-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: prod-comms-consumer-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: prod-comms-consumer-v1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-custom-32-65536 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/search-relay-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/search-relay-cc.yaml deleted file mode 100644 index 21cb1be..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/search-relay-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: search-relay -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: search-relay - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/sumoduo-azul-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/sumoduo-azul-cc.yaml deleted file mode 100644 index 79bbff4..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/sumoduo-azul-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumoduo-azul -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumoduo-azul - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3-highcpu-44 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - - location: - zones: - - asia-southeast1-c - machineType: c3-highcpu-44 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - - location: - zones: - - asia-southeast1-b - machineType: c3-highcpu-44 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/sumoduo-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/sumoduo-cc.yaml deleted file mode 100644 index 58fa23e..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/sumoduo-cc.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumoduo -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumoduo - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3-highcpu-44 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-a - machineType: c4-highcpu-48 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: c3-highcpu-44 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-c - machineType: c4-highcpu-48 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: c3-highcpu-44 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - location: - zones: - - asia-southeast1-b - machineType: c4-highcpu-48 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/sumoduolite-azul-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/sumoduolite-azul-cc.yaml deleted file mode 100644 index 619a695..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/sumoduolite-azul-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumoduolite-azul -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumoduolite-azul - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/sumoduolite-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/sumoduolite-cc.yaml deleted file mode 100644 index d465f58..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/sumoduolite-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumoduolite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumoduolite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4-highcpu-32 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4-highcpu-32 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4-highcpu-32 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/sumoocta-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/sumoocta-cc.yaml deleted file mode 100644 index 359fe51..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/sumoocta-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumoocta -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumoocta - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3-highmem-44 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - - location: - zones: - - asia-southeast1-c - machineType: c3-highmem-44 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - - location: - zones: - - asia-southeast1-b - machineType: c3-highmem-44 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/sumotetra-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/sumotetra-cc.yaml deleted file mode 100644 index 2edf1be..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/sumotetra-cc.yaml +++ /dev/null @@ -1,45 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumotetra -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumotetra - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/sumotetralite-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/sumotetralite-cc.yaml deleted file mode 100644 index a5aa359..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/sumotetralite-cc.yaml +++ /dev/null @@ -1,72 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumotetralite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumotetralite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n2-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - - location: - zones: - - asia-southeast1-a - machineType: n4-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-c - machineType: n2-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - - location: - zones: - - asia-southeast1-c - machineType: n4-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - location: - zones: - - asia-southeast1-b - machineType: n2-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - - location: - zones: - - asia-southeast1-b - machineType: n4-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/vm-agent-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/vm-agent-cc.yaml deleted file mode 100644 index 0266c95..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/vm-agent-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vm-agent -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vm-agent - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4-highcpu-144 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/vmagent-c4d-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/vmagent-c4d-cc.yaml deleted file mode 100644 index 0f396c5..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/vmagent-c4d-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmagent-c4d -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmagent-c4d - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4d-highcpu-192 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/vminsert-mds-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/vminsert-mds-cc.yaml deleted file mode 100644 index 0dd65d5..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/vminsert-mds-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vminsert-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vminsert-mds - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-highcpu-32 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/vmselect-mds-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/vmselect-mds-cc.yaml deleted file mode 100644 index a5db7e0..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/vmselect-mds-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmselect-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmselect-mds - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-highcpu-44 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml b/helm-overrides/gke-demand-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml deleted file mode 100644 index 92ebc9c..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage-n4d -spec: - nodePoolConfig: - serviceAccount: sa-common-np-dmnd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmstorage-n4d - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highmem-96 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-demand-prd-ase1a/conntrack-adjuster/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 6824ce1..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,18 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - max: 2097152 - hashsize: 524288 - sleepInterval: 30 - -nodeAffinity: - enabled: true - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: In - values: - - "contour-external" - - "contour-internal-0" - - "contour-intra-0" diff --git a/helm-overrides/gke-demand-prd-ase1a/contour-ca-issuer/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/contour-ca-issuer/custom-values.yaml deleted file mode 100644 index 45328b0..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/contour-ca-issuer/custom-values.yaml +++ /dev/null @@ -1,17 +0,0 @@ -# ClusterIssuer override for gke-demand-prd-ase1a (prd demand cluster, ase1a zone). -# ClusterIssuer is a cluster-scoped resource — this file pins the issuer -# identity for this cluster so the name is auditable per-cluster. -# -# Must match the `issuerRef.name` in consuming Certificate CRs (see -# devops-helm-charts/2.0.0/templates/proxyless-grpc-cert.yaml). - -issuerName: contour-demand-prd-ca-issuer -rootCASecretName: contour-demand-ca - -externalSecret: - enabled: true - vaultPath: meesho/devops/contour/root-ca - refreshInterval: "0" - secretStoreRef: - name: vault-backend -namespace: cert-manager-demand-prd \ No newline at end of file diff --git a/helm-overrides/gke-demand-prd-ase1a/contour-cert-checker/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/contour-cert-checker/custom-values.yaml deleted file mode 100644 index 9a0b5d6..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=gke-demand-prd-ase1a"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - cloud.google.com/compute-class: demand-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: demand-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/gke-demand-prd-ase1a/contour-external/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/contour-external/custom-values.yaml deleted file mode 100644 index 95f438c..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/contour-external/custom-values.yaml +++ /dev/null @@ -1,111 +0,0 @@ -fullnameOverride: "contour-external-demand-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - network: - num-trusted-hops: 1 -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: demand - team: demand-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-shared-cc-v1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: demand - team: demand-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-external-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-external-cc-v1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "40" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 2' - resources: - requests: - cpu: 2 - memory: 512Mi - limits: - cpu: 2 - memory: 4Gi - service: - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-demand-a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-demand-prd-ase1a/contour-internal-0/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/contour-internal-0/custom-values.yaml deleted file mode 100644 index 9feca8f..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,120 +0,0 @@ -fullnameOverride: "contour-internal-0-demand-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: demand - team: demand-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 4Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-shared-cc-v1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - updateStrategy: - rollingUpdate: - maxSurge: 10% - enabled: true - podLabels: - bu: demand - team: demand-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-0-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-internal-0-cc-v1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "55" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 14' - resources: - requests: - cpu: 14 - memory: 8Gi - limits: - cpu: 14 - memory: 27Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-demand-a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-demand-prd-ase1a/contour-internal-1/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/contour-internal-1/custom-values.yaml deleted file mode 100644 index 08c4227..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,117 +0,0 @@ -fullnameOverride: "contour-internal-1-demand-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: demand - team: demand-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 8 - memory: 10Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-shared-cc-v1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: demand - team: demand-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-1-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-internal-1-cc-v1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 6' - resources: - requests: - cpu: 6 - memory: 3Gi - limits: - cpu: 6 - memory: 12Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int1-demand-a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-demand-prd-ase1a/contour-internal-intra-0/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index 36d7eb6..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,118 +0,0 @@ -fullnameOverride: "contour-internal-0-demand-intra-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: demand - team: demand-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 8 - memory: 10Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-shared-cc-v1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - updateStrategy: - rollingUpdate: - maxSurge: 10% - enabled: true - podLabels: - bu: demand - team: demand-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-intra-0-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-intra-0-cc-v1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "55" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 14' - resources: - requests: - cpu: 14 - memory: 8Gi - limits: - cpu: 14 - memory: 27Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-demand-prd-ase1a/contour-internal-intra-1/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index 228082a..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,115 +0,0 @@ -fullnameOverride: "contour-internal-1-demand-intra-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: demand - team: demand-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 12 - memory: 22Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-shared-cc-v1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: demand - team: demand-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-intra-1-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-intra-1-cc-v1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 6' - resources: - requests: - cpu: 6 - memory: 3Gi - limits: - cpu: 6 - memory: 12Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-demand-prd-ase1a/coredns/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/coredns/custom-values.yaml deleted file mode 100644 index 305d0b1..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/coredns/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -replicaCount: 72 -communicationType: "intra" - -labels: - bu: demand - team: demand-devops - env: prd - -clusterIP: 10.1.0.2 - - -resources: - limits: - cpu: 150m - memory: 128Mi - requests: - cpu: 150m - memory: 128Mi - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "demand-devops" - effect: "NoSchedule" - -nodeSelector: - cloud.google.com/compute-class: demand-devops - kubernetes.io/os: linux diff --git a/helm-overrides/gke-demand-prd-ase1a/coroot-node-agent/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/coroot-node-agent/custom-values.yaml deleted file mode 100644 index c3f8d74..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/coroot-node-agent/custom-values.yaml +++ /dev/null @@ -1,52 +0,0 @@ -fullnameOverride: "coroot-node-agent-demand-prd" -priorityClassName: "low-priority" - -scrape: "false" # disable scraping for now - -resources: - requests: - cpu: "300m" - memory: "1Gi" - limits: - cpu: "500m" - memory: "1.5Gi" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-demand-contour-ext - operator: NotIn - values: - - dedicated - - key: p-demand-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-demand-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-demand-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-demand-default-prd-ase1 - - key: cloud.google.com/compute-class - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external \ No newline at end of file diff --git a/helm-overrides/gke-demand-prd-ase1a/external-secrets/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/external-secrets/custom-values.yaml deleted file mode 100644 index b8e4927..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/external-secrets/custom-values.yaml +++ /dev/null @@ -1,61 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Annotations to add to Pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: demand-devops - nodeSelector: - cloud.google.com/compute-class: demand-devops - certController: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: demand-devops - nodeSelector: - cloud.google.com/compute-class: demand-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - webhook: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: demand-devops - nodeSelector: - cloud.google.com/compute-class: demand-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" \ No newline at end of file diff --git a/helm-overrides/gke-demand-prd-ase1a/flagger/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/flagger/custom-values.yaml deleted file mode 100644 index 12b50ff..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/flagger/custom-values.yaml +++ /dev/null @@ -1,62 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - cloud.google.com/compute-class: demand-devops - -tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: demand-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: demand-devops - bu: infra - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-demand-rollout-service.prd-demand-rollout-service.svc.cluster.local" - -rolloutService: - enabled: true - apps: \ No newline at end of file diff --git a/helm-overrides/gke-demand-prd-ase1a/fluentd/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/fluentd/custom-values.yaml deleted file mode 100644 index d3ab318..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/fluentd/custom-values.yaml +++ /dev/null @@ -1,734 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-dmnd-dmsre-fluentd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: demand - team: sre - type: fluentd - service: fluentd-demand-a-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: demand - team: sre - type: fluentd - service: fluentd-demand-a-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -# - secretRef: -# name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: - - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - from_encoding "utf-8" - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/gke-demand-prd-ase1a/keda/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/keda/custom-values.yaml deleted file mode 100644 index 3378991..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/keda/custom-values.yaml +++ /dev/null @@ -1,61 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - cloud.google.com/compute-class: demand-devops - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "demand-devops" - effect: "NoSchedule" - podLabels: - bu: "demand" - team: "demand-devops" - metricsAdapter: - bu: "demand" - team: "demand-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 1600Mi - requests: - cpu: 500m - memory: 850Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 60m - memory: 200Mi - metricServer: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 200m - memory: 200Mi - podAnnotations: - # -- Pod annotations for KEDA operator - keda: - prometheus.io/path: /metrics - prometheus.io/port: '8080' - prometheus.io/scrape: 'true' - # -- Pod annotations for KEDA Metrics Adapter - metricsAdapter: - prometheus.io/path: /metrics - prometheus.io/port: '8080' - prometheus.io/scrape: 'true' - # -- Pod annotations for KEDA Admission webhooks - webhooks: - prometheus.io/path: /metrics - prometheus.io/port: '8080' - prometheus.io/scrape: 'true' - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' diff --git a/helm-overrides/gke-demand-prd-ase1a/kube-dns/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/kube-dns/custom-values.yaml deleted file mode 100644 index bc8be09..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.1.0.2"],"prd.mrouter.int.svc.cluster.local":["10.1.0.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.1.0.2"]} diff --git a/helm-overrides/gke-demand-prd-ase1a/kube-events/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/kube-events/custom-values.yaml deleted file mode 100644 index e370ba3..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-demand-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - cloud.google.com/compute-class: demand-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: demand-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: demand-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: demand-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: demand-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: demand-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/gke-demand-prd-ase1a/kube-state-metrics/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/kube-state-metrics/custom-values.yaml deleted file mode 100644 index cd8d87c..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,429 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: "" - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-demand-a-prd - -imagePullSecrets: [] -# - name: "image-pull-secret" - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "demand" - team: "demand-sre" - service: "kube-state-metrics-demand-a-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - cloud.google.com/compute-class: demand-devops - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: demand-devops - operator: Equal - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 350m - memory: 1.5Gi - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/gke-demand-prd-ase1a/kubectl-mcp-server/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/kubectl-mcp-server/custom-values.yaml deleted file mode 100644 index 5c349e4..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/kubectl-mcp-server/custom-values.yaml +++ /dev/null @@ -1,119 +0,0 @@ -fullnameOverride: "kubectl-mcp-server" - -replicas: 1 - -image: - # In-house hardened build (helm-templates/kubectl-mcp-server/docker/) — - # NOT the upstream Docker Hub image. 266→18 HIGH/CRIT vulns, non-root, - # multi-stage slim base, kubectl v1.33.12 / helm v3.21.0. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/prd/devop/kubectl-mcp-server - tag: "v2" - pullPolicy: IfNotPresent - -labels: - bu: infra - team: devops - service: kubectl-mcp-server - env: prd - -serviceAccount: - create: true - annotations: {} - -rbac: - create: true - -podAnnotations: {} - -podSecurityContext: {} - -securityContext: - runAsNonRoot: true - runAsUser: 1000 - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -priorityClassName: "" - -nodeSelector: - cloud.google.com/compute-class: "demand-devops" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "demand-devops" - effect: NoSchedule - -affinity: {} - -resources: - requests: - cpu: 250m - memory: 512Mi - limits: - cpu: 1000m - memory: 1Gi - -mcp: - mode: single - transport: http - host: "0.0.0.0" - port: 8000 - -auth: - allowAnonymous: false - -externalSecrets: - enabled: true - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/prd/cntr/devop/kubectl-mcp-server-demand" - -# tcpSocket on purpose — streamable-http transport exposes only /mcp, -# there is no /health route (see chart values.yaml note). -livenessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - -readinessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - -service: - port: 8000 - type: ClusterIP - -# Contour standard (2.0.0-style): renders HTTPProxy parent/child + -intra -# instead of a networking.k8s.io Ingress. -createContourGateway: true - -ingress: - enabled: true - # demand prd exposes the -0 contour internal classes (no -1); the chart - # auto-derives contour-internal-intra-0. Verified on gke-demand-prd-ase1a. - ingressClassName: contour-internal-0 - servicePortNumber: 8000 - hosts: - - host: kubectl-mcp-server-demand.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 diff --git a/helm-overrides/gke-demand-prd-ase1a/kyverno/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/kyverno/custom-values.yaml deleted file mode 100644 index 1563315..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/kyverno/custom-values.yaml +++ /dev/null @@ -1,2244 +0,0 @@ -global: - - # -- Internal settings used with `helm template` to generate install manifest - # @ignored - templating: - enabled: false - debug: false - version: ~ - - image: - # -- (string) Global value that allows to set a single image registry across all deployments. - # When set, it will override any values set under `.image.registry` across the chart. - registry: ~ - # -- (list) Global list of Image pull secrets - # When set, it will override any values set under `imagePullSecrets` under different components across the chart. - imagePullSecrets: [] - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - caCertificates: - # -- Global CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - # Individual controller values will override this global value - data: ~ - - # -- Global value to set single volume to be mounted for CA certificates for all deployments. - # Not used when `.Values.global.caCertificates.data` is defined - # Individual controller values will override this global value - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Additional container environment variables to apply to all containers and init containers - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Global node labels for pod assignment. Non-global values will override the global value. - nodeSelector: - cloud.google.com/compute-class: demand-kyverno - - # -- Global List of node taints to tolerate. Non-global values will override the global value. - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "demand-kyverno" - effect: "NoSchedule" - -# -- (string) Override the name of the chart -nameOverride: ~ - -# -- (string) Override the expanded name of the chart -fullnameOverride: ~ - -# -- (string) Override the namespace the chart deploys to -namespaceOverride: ~ - -upgrade: - # -- Upgrading from v2 to v3 is not allowed by default, set this to true once changes have been reviewed. - fromV2: false - -apiVersionOverride: - # -- (string) Override api version used to create `PodDisruptionBudget`` resources. - # When not specified the chart will check if `policy/v1/PodDisruptionBudget` is available to - # determine the api version automatically. - podDisruptionBudget: ~ - -rbac: - roles: - # -- Aggregate ClusterRoles to Kubernetes default user-facing roles. For more information, see [User-facing roles](https://kubernetes.io/docs/reference/access-authn-authz/rbac/#user-facing-roles) - aggregate: - admin: true - view: true - -# Use openreports.io as the API group for reporting -openreports: - # -- Enable OpenReports feature in controllers - enabled: false - # -- Whether to install CRDs from the upstream OpenReports chart. Setting this to true requires enabled to also be true. - installCrds: false - -# CRDs configuration -crds: - - # -- Whether to have Helm install the Kyverno CRDs, if the CRDs are not installed by Helm, they must be added before policies can be created - install: true - - reportsServer: - # -- Kyverno reports-server is used in your cluster - enabled: false - - groups: - - # -- Install CRDs in group `kyverno.io` - kyverno: - cleanuppolicies: true - clustercleanuppolicies: true - clusterpolicies: true - globalcontextentries: true - policies: true - policyexceptions: true - updaterequests: true - - # -- Install CRDs in group `policies.kyverno.io` - policies: - validatingpolicies: true - policyexceptions: true - imagevalidatingpolicies: true - namespacedimagevalidatingpolicies: true - mutatingpolicies: true - generatingpolicies: true - deletingpolicies: true - namespaceddeletingpolicies: true - namespacedvalidatingpolicies: true - - # -- Install CRDs in group `reports.kyverno.io` - reports: - clusterephemeralreports: true - ephemeralreports: true - - # -- Install CRDs in group `wgpolicyk8s.io` - wgpolicyk8s: - clusterpolicyreports: false ##lk - policyreports: false ##lk - - # -- Additional CRDs annotations (Replace=true avoids 262144-byte annotation limit on large CRDs) - annotations: - argocd.argoproj.io/sync-options: Replace=true - # strategy.spinnaker.io/replace: 'true' - - # -- Additional CRDs labels - customLabels: {} - - migration: - - # -- Enable CRDs migration using helm post upgrade hook - enabled: true - - # -- Resources to migrate - resources: - - cleanuppolicies.kyverno.io - - clustercleanuppolicies.kyverno.io - - clusterpolicies.kyverno.io - - globalcontextentries.kyverno.io - - policies.kyverno.io - - policyexceptions.kyverno.io - - updaterequests.kyverno.io - - deletingpolicies.policies.kyverno.io - - generatingpolicies.policies.kyverno.io - - imagevalidatingpolicies.policies.kyverno.io - - namespacedimagevalidatingpolicies.policies.kyverno.io - - mutatingpolicies.policies.kyverno.io - - namespaceddeletingpolicies.policies.kyverno.io - - namespacedvalidatingpolicies.policies.kyverno.io - - policyexceptions.policies.kyverno.io - - validatingpolicies.policies.kyverno.io - - image: - # -- (string) Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- (string) Image repository - repository: kyverno/kyverno-cli - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- (string) Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podResources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -# Configuration -config: - - # -- Create the configmap. - create: true - - # -- Preserve the configmap settings during upgrade. - preserve: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - # -- Enable registry mutation for container images. Enabled by default. - enableDefaultRegistryMutation: true - - # -- The registry hostname used for the image mutation. - defaultRegistry: docker.io - - # -- Exclude groups - excludeGroups: - - system:nodes - - # -- Exclude usernames - excludeUsernames: [] - # - '!system:kube-scheduler' - - # -- Exclude roles - excludeRoles: [] - - # -- Exclude roles - excludeClusterRoles: [] - - # -- Generate success events. - generateSuccessEvents: false - - # -- Resource types to be skipped by the Kyverno policy engine. - # Make sure to surround each entry in quotes so that it doesn't get parsed as a nested YAML list. - # These are joined together without spaces, run through `tpl`, and the result is set in the config map. - # @default -- See [values.yaml](values.yaml) - resourceFilters: - - '[Event,*,*]' - # - '[*/*,kube-system,*]' - - '[*/*,kube-public,*]' - - '[*/*,kube-node-lease,*]' - - '[Node,*,*]' - - '[Node/?*,*,*]' - - '[APIService,*,*]' - - '[APIService/?*,*,*]' - - '[TokenReview,*,*]' - - '[SubjectAccessReview,*,*]' - - '[SelfSubjectAccessReview,*,*]' - - '[Binding,*,*]' - - '[Pod/binding,*,*]' - - '[ReplicaSet,*,*]' - - '[ReplicaSet/?*,*,*]' - - '[EphemeralReport,*,*]' - - '[ClusterEphemeralReport,*,*]' - # exclude resources from the chart - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:additional]' - - '[ClusterRoleBinding,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.configMapName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.metricsConfigMapName" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Job,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[Job/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.admission-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.background-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - # -- Sets the threshold for the total number of UpdateRequests generated for mutateExisitng and generate policies. - updateRequestThreshold: 1000 - - # -- Defines the `namespaceSelector`/`objectSelector` in the webhook configurations. - # The Kyverno namespace is excluded if `excludeKyvernoNamespace` is `true` (default) - webhooks: - # Exclude namespaces - # namespaceSelector: lk-s2 - # matchExpressions: - # - key: kubernetes.io/metadata.name - # operator: NotIn - # values: - # - kube-systemm lk-s2 - # Exclude objects - # objectSelector: - # matchExpressions: - # - key: webhooks.kyverno.io/exclude - # operator: DoesNotExist - - # -- Defines annotations to set on webhook configurations. - webhookAnnotations: - # Example to disable admission enforcer on AKS: - # 'admissions.enforcer/disabled': 'true' lk-s3 - - # -- Defines labels to set on webhook configurations. - webhookLabels: {} - # Example to adopt webhook resources in ArgoCD: - # 'argocd.argoproj.io/instance': 'kyverno' - - # -- Defines match conditions to set on webhook configurations (requires Kubernetes 1.27+). - matchConditions: [] - - # -- Exclude Kyverno namespace - # Determines if default Kyverno namespace exclusion is enabled for webhooks and resourceFilters - excludeKyvernoNamespace: true - - # -- resourceFilter namespace exclude - # Namespaces to exclude from the default resourceFilters - resourceFiltersExcludeNamespaces: [] - - # -- resourceFilters exclude list - # Items to exclude from config.resourceFilters - resourceFiltersExclude: [] - - # -- resourceFilter namespace include - # Namespaces to include to the default resourceFilters - resourceFiltersIncludeNamespaces: [] - - # -- resourceFilters include list - # Items to include to config.resourceFilters - resourceFiltersInclude: [] - -# Metrics configuration -metricsConfig: - - # -- Create the configmap. - create: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - namespaces: - - # -- List of namespaces to capture metrics for. - include: [] - - # -- list of namespaces to NOT capture metrics for. - exclude: [] - - # -- (string) Rate at which metrics should reset so as to clean up the memory footprint of kyverno metrics, if you might be expecting high memory footprint of Kyverno's metrics. Default: 0, no refresh of metrics. WARNING: This flag is not working since Kyverno 1.8.0 - metricsRefreshInterval: ~ - # metricsRefreshInterval: 24h - - # -- (list) Configures the bucket boundaries for all Histogram metrics, changing this configuration requires restart of the kyverno admission controller - bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 15, 20, 25, 30] - - # -- (map) Configures the exposure of individual metrics, by default all metrics and all labels are exported, changing this configuration requires restart of the kyverno admission controller - metricsExposure: - kyverno_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_image_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_mutating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_generating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_admission_review_duration_seconds: - # enabled: false - disabledLabelDimensions: ["resource_namespace"] - kyverno_policy_rule_info_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_policy_results_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_admission_requests_total: - disabledLabelDimensions: ["resource_namespace"] - kyverno_cleanup_controller_deletedobjects_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - -# -- Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -imagePullSecrets: {} - # regcred: - # registry: foo.example.com - # username: foobar - # password: secret - # regcred2: - # registry: bar.example.com - # username: barbaz - # password: secret2 - -# -- Existing Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -existingImagePullSecrets: [] - # - test-registry - # - other-test-registry - -# Tests configuration -test: - # -- Sleep time before running test - sleep: 20 - - image: - # -- (string) Image registry - registry: curlimages - # -- Image repository - repository: curl - # -- Image tag - # Defaults to `latest` if omitted - tag: '8.10.1' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - # -- Security context for the test containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- Additional Pod annotations - podAnnotations: {} - - # -- List of node taints to tolerate - tolerations: [] - -# -- Additional labels -customLabels: {} - - -webhooksCleanup: - # -- Create a helm pre-delete hook to cleanup webhooks. - enabled: true - - autoDeleteWebhooks: - # -- Allow webhooks controller to delete webhooks using finalizers - enabled: false - - image: - # -- (string) Image registry - registry: registry.k8s.io - # -- Image repository - repository: kubectl - # -- Image tag - # Defaults to `latest` if omitted - tag: 'v1.32.7' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -grafana: - # -- Enable grafana dashboard creation. - enabled: false - - # -- Configmap name template. - configMapName: '{{ include "kyverno.fullname" . }}-grafana' - - # -- (string) Namespace to create the grafana dashboard configmap. - # If not set, it will be created in the same namespace where the chart is deployed. - namespace: ~ - - # -- Grafana dashboard configmap annotations. - annotations: {} - - # -- Grafana dashboard configmap labels - labels: - grafana_dashboard: "1" - - # -- create GrafanaDashboard custom resource referencing to the configMap. - # according to https://grafana-operator.github.io/grafana-operator/docs/examples/dashboard_from_configmap/readme/ - grafanaDashboard: - create: false - folder: kyverno - allowCrossNamespaceImport: true - matchLabels: - dashboards: "grafana" - -# Features configuration -features: - admissionReports: - # -- Enables the feature - enabled: true - aggregateReports: - # -- Enables the feature - enabled: true - policyReports: - # -- Enables the feature - enabled: false ##lk - validatingAdmissionPolicyReports: - # -- Enables the feature - enabled: true - mutatingAdmissionPolicyReports: - # -- Enables the feature - enabled: false - reporting: - # -- Enables the feature - validate: true - # -- Enables the feature - mutate: true - # -- Enables the feature - mutateExisting: true - # -- Enables the feature - imageVerify: true - # -- Enables the feature - generate: true - autoUpdateWebhooks: - # -- Enables the feature - enabled: true - backgroundScan: - # -- Enables the feature - enabled: true - # -- Number of background scan workers - backgroundScanWorkers: 2 - # -- Background scan interval - backgroundScanInterval: 1h - # -- Skips resource filters in background scan - skipResourceFilters: true - configMapCaching: - # -- Enables the feature - enabled: true - controllerRuntimeMetrics: - # -- Bind address for controller-runtime metrics (use "0" to disable it) - bindAddress: ":8080" - deferredLoading: - # -- Enables the feature - enabled: true - dumpPayload: - # -- Enables the feature - enabled: false - forceFailurePolicyIgnore: - # -- Enables the feature - enabled: false - generateValidatingAdmissionPolicy: - # -- Enables the feature - enabled: true - generateMutatingAdmissionPolicy: - # -- Enables the feature - enabled: false - dumpPatches: - # -- Enables the feature - enabled: false - globalContext: - # -- Maximum allowed response size from API Calls. A value of 0 bypasses checks (not recommended) - maxApiCallResponseLength: 2000000 - logging: - # -- Logging format - format: text - # -- Logging verbosity - verbosity: 2 - omitEvents: - # -- Events which should not be emitted (possible values `PolicyViolation`, `PolicyApplied`, `PolicyError`, and `PolicySkipped`) - eventTypes: - - PolicyApplied - - PolicySkipped - # - PolicyViolation - # - PolicyError - policyExceptions: - # -- Enables the feature - enabled: true - # -- Restrict policy exceptions to a single namespace - # Set to "*" to allow exceptions in all namespaces - namespace: '' - protectManagedResources: - # -- Enables the feature - enabled: false - registryClient: - # -- Allow insecure registry - allowInsecure: false - # -- Enable registry client helpers - credentialHelpers: - - default - - google - - amazon - - azure - - github - ttlController: - # -- Reconciliation interval for the label based cleanup manager - reconciliationInterval: 1m - tuf: - # -- Enables the feature - enabled: false - # -- (string) Path to Tuf root - root: ~ - # -- (string) Raw Tuf root - rootRaw: ~ - # -- (string) Tuf mirror - mirror: ~ - -# Admission controller configuration -admissionController: - autoscaling: - # -- Enable horizontal pod autoscaling - enabled: true - - # -- Minimum number of pods - minReplicas: 1 - - # -- Maximum number of pods - maxReplicas: 10 - - # -- Target CPU utilization percentage - targetCPUUtilizationPercentage: 80 - - # -- Configurable scaling behavior - behavior: {} - - # -- Overrides features defined at the root level - featuresOverride: - admissionReports: - # -- Max number of admission reports allowed in flight until the admission controller stops creating new ones - backPressureThreshold: 1000 - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- The ServiceAccount name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno admission controller activities. - # This will help ensure Kyverno stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- admissionController webhook server port - # in case you are using hostNetwork: true, you might want to change the port the webhookServer is listening to - webhookServer: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - admission-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.admissionController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - initContainer: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyvernopre - # -- (string) Image tag - # If missing, defaults to image.tag - tag: ~ - # -- (string) Image pull policy - # If missing, defaults to image.pullPolicy - pullPolicy: ~ - - resources: - # -- Pod resource limits - limits: - cpu: 1000m - memory: 1Gi - # -- Pod resource requests - requests: - cpu: 1000m - memory: 1Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - container: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyverno - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - resources: - # -- Pod resource limits - limits: - cpu: 1900m - memory: 3Gi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Array of extra init containers - extraInitContainers: [] - # - name: init-container - # image: busybox - # command: ['sh', '-c', 'echo Hello'] - - # -- Array of extra containers to run alongside kyverno - extraContainers: [] - # - name: myapp-container - # image: busybox - # command: ['sh', '-c', 'echo Hello && sleep 3600'] - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Kyverno's metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Background controller configuration -backgroundController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable background controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: - - apiGroups: - - networking.k8s.io - resources: - - ingresses - - ingressclasses - - networkpolicies - verbs: - - create - - update - - patch - - delete - - apiGroups: - - rbac.authorization.k8s.io - resources: - - rolebindings - - roles - verbs: - - create - - update - - patch - - delete - - apiGroups: - - '' - resources: - - configmaps - - resourcequotas - - limitranges - verbs: - - create - - update - - patch - - delete - - apiGroups: - - resource.k8s.io - resources: - - resourceclaims - - resourceclaimtemplates - verbs: - - create - - delete - - update - - patch - - deletecollection - - apiGroups: - - autoscaling - resources: - - horizontalpodautoscalers - verbs: - - get - - list - - watch - - update - - patch - - apiGroups: - - apps - resources: - - deployments - - statefulsets - - replicasets - verbs: - - get - - list - - watch - - update - - patch - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - # - patch - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/background-controller - # -- Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 1900m - memory: 3Gi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - background-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.backgroundController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - # -- backgroundController server port - # in case you are using hostNetwork: true, you might want to change the port the backgroundController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Cleanup controller configuration -cleanupController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable cleanup controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - delete - # - list - # - watch - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/cleanup-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- cleanupController server port - # in case you are using hostNetwork: true, you might want to change the port the cleanupController is listening to - server: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 1100m - memory: 1.2Gi - # -- Pod resource requests - requests: - cpu: 1100m - memory: 1.2Gi - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - cleanup-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `service.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Reports controller configuration -reportsController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable reports controller. - enabled: false ##lk - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/reports-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno reports controller activities. - # This will help ensure Kyverno reports stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 200m - memory: 128Mi - # -- Pod resource requests - requests: - cpu: 100m - memory: 64Mi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - reports-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.reportsController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- (string) Service node port. - # Only used if `type` is `NodePort`. - nodePort: ~ - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- (string) Traces receiver address - address: ~ - # -- (string) Traces receiver port - port: ~ - # -- (string) Traces receiver credentials - creds: ~ - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- (string) Otel collector endpoint - collector: ~ - # -- (string) Otel collector credentials - creds: ~ - - # -- reportsController server port - # in case you are using hostNetwork: true, you might want to change the port the reportsController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - - # -- Enable sanity check for reports CRDs - sanityChecks: true diff --git a/helm-overrides/gke-demand-prd-ase1a/kyverno/policies/protect-namespace.yaml b/helm-overrides/gke-demand-prd-ase1a/kyverno/policies/protect-namespace.yaml deleted file mode 100644 index 460e19d..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/kyverno/policies/protect-namespace.yaml +++ /dev/null @@ -1,40 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: protect-critical-namespaces -spec: - background: false - rules: - - name: deny-deletion-of-critical-namespaces - match: - any: - - resources: - kinds: - - Namespace - names: - - keda-demand-prd - - contour-internal-0-demand-prd - - contour-internal-0-demand-prd-intra - - contour-external-demand-prd - - external-secrets-demand-prd - - flagger-demand-prd - - victoriametrics - - kube-system - - monitoring - - telegraf-operator - - loadtester - - opentelemetry - - fluentd - - kube-events - - observability - - contour-cert-checker-ns - - aurva-dataplane - validate: - failureAction: Enforce - message: "Deletion of critical namespaces is not allowed." - deny: - conditions: - any: - - key: "{{request.operation}}" - operator: Equals - value: DELETE diff --git a/helm-overrides/gke-demand-prd-ase1a/kyverno/policies/restrict-replicas.yaml b/helm-overrides/gke-demand-prd-ase1a/kyverno/policies/restrict-replicas.yaml deleted file mode 100644 index 101bbeb..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/kyverno/policies/restrict-replicas.yaml +++ /dev/null @@ -1,29 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: restrict-minimum-replicas-selected-ns -spec: - background: false - rules: - - name: min-replicas-deployments-statefulsets - match: - resources: - kinds: - - Deployment - - StatefulSet - - Deployment/scale - - StatefulSet/scale - names: - - coredns - namespaces: - - kube-system - operations: - - CREATE - - UPDATE - validate: - failureAction: Enforce - message: "Deployments and StatefulSets must have at least 5 replicas in this namespace." - anyPattern: - # Case 1: replicas explicitly set and >= 5 - - spec: - replicas: ">=5" diff --git a/helm-overrides/gke-demand-prd-ase1a/loadtester/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/loadtester/custom-values.yaml deleted file mode 100644 index 40edb8d..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/loadtester/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -replicaCount: 1 - -Namespace: loadtester - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/dev/devops/flagger-loadtester - tag: 0.30.0 - pullPolicy: IfNotPresent - pullSecret: - -podLabels: - bu: demand - team: demand-shared - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - -podPriorityClassName: "" - -logLevel: info -cmd: - timeout: 1h - namespaceRegexp: "" - -nameOverride: "loadtester" -fullnameOverride: "" - -env: [] - -service: - type: ClusterIP - port: 80 - -resources: - requests: - cpu: 10m - memory: 64Mi - -volumes: [] -volumeMounts: [] - -nodeSelector: - cloud.google.com/compute-class: demand-devops - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "demand-devops" - effect: "NoSchedule" - -affinity: {} - -ingress: - slowStart: - enabled: false - enabled: true - ingressClassName: contour-internal-0 - servicePort: http - hosts: - - host: demand-flagger-loadtester.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / -createContourGateway: true -contourResponseTimeout: false -rbac: - # rbac.create: `true` if rbac resources should be created - create: true - # rbac.scope: `cluster` to create cluster-scope rbac resources (ClusterRole/ClusterRoleBinding) - # otherwise, namespace-scope rbac resources will be created (Role/RoleBinding) - scope: - # rbac.rules: array of rules to apply to the role. example: - # rules: - # - apiGroups: [""] - # resources: ["pods"] - # verbs: ["list", "get"] - rules: [] - -# name of an existing service account to use - if not creating rbac resources -serviceAccountName: "" - -# App Mesh virtual node settings (to be used for AppMesh v1beta1) -meshName: "" -#backends: -# - app1.namespace -# - app2.namespace - -# App Mesh virtual node settings (to be used for AppMesh v1beta2) -appmesh: - enabled: false - backends: - - podinfo - - podinfo-canary - -#Istio virtual service and gatway settings. TLS secrets should be in namespace before enbaled it. ( secret format loadtester.fullname ) -istio: - enabled: false - host: flagger-loadtester.flagger - gateway: - enabled: false - tls: - enabled: false - httpsRedirect: false - -# when enabled, it will add a security context for the loadtester pod -securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 100 - runAsGroup: 101 - -podDisruptionBudget: - enabled: false - minAvailable: 1 diff --git a/helm-overrides/gke-demand-prd-ase1a/node-thp-config/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/node-thp-config/custom-values.yaml deleted file mode 100644 index 9b4dc44..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/node-thp-config/custom-values.yaml +++ /dev/null @@ -1,9 +0,0 @@ -daemonSet: - namespace: prd-node-thp-config - -baseMatchExpressions: -- key: dedicated - operator: In - values: - - "sumoduo-azul" - - "sumoduolite-azul" diff --git a/helm-overrides/gke-demand-prd-ase1a/opentelemetry-coralogix/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/opentelemetry-coralogix/custom-values.yaml deleted file mode 100644 index 9b86ec3..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/opentelemetry-coralogix/custom-values.yaml +++ /dev/null @@ -1,281 +0,0 @@ -global: - domain: "coralogixsg.com" - defaultApplicationName: "default" - defaultSubsystemName: "nodes" - - # Old endpoint based configuration, - # please use domain instead. - traces: - endpoint: "" - metrics: - endpoint: "" - logs: - endpoint: "" - -# set distribution to openshift for openshift clusters -distribution: "" - -# Opentelemetry collector configuration -dedicatedValue: false -schedulerName: default-scheduler - -labels: - bu: "demand" - team: "sre" - service: "opentelemetry-demand-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: false - key: prd/admin/coralogix-keys - secretStoreRef: - name: vault-backend - -image: - # If you want to use the core image `otel/opentelemetry-collector`, you also need to change `command.name` value to `otelcol`. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - # repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - tag: "0.153.0" - # When digest is set to a non-empty value, images will be pulled by digest (regardless of tag value). - digest: "" -mode: daemonset -rollout: - rollingUpdate: - # When 'mode: daemonset', maxSurge cannot be used when hostPort is set for any of the ports - # maxSurge: 25% - maxUnavailable: 1 - strategy: RollingUpdate -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" -hostNetwork: true -dnsPolicy: "ClusterFirstWithHostNet" -fullnameOverride: opentelemetry-demand-prd - -presets: - logsCollection: - enabled: false - storeCheckpoints: false - maxRecombineLogSize: 1048576 - extraFilelogOperators: [] -# - type: recombine -# combine_field: body -# source_identifier: attributes["log.file.path"] -# is_first_entry: body matches "^(YOUR-LOGS-REGEX)" - kubernetesAttributes: - enabled: false - hostMetrics: - enabled: false - kubeletMetrics: - enabled: false - -extraEnvs: -- name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" -- name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName -config: - extensions: {} - # zpages: - # endpoint: localhost:55679 - # pprof: - # endpoint: localhost:1777 - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 4s - tls: - insecure: true - sending_queue: - enabled: true - queue_size: 6000 - num_consumers: 200 - resolver: - dns: - hostname: opentelemetry-admin-prd.opentelemetry.svc.clusterset.local - # coralogix: - # timeout: "30s" - # private_key: "${CORALOGIX_PRIVATE_KEY}" - # domain: "{{.Values.global.domain}}" - # traces: - # endpoint: "{{ .Values.global.traces.endpoint }}" - # metrics: - # endpoint: "{{ .Values.global.metrics.endpoint }}" - # logs: - # endpoint: "{{ .Values.global.logs.endpoint }}" - # application_name_attributes: - # - "k8s.namespace.name" - # - "service.namespace" - # subsystem_name_attributes: - # - "k8s.deployment.name" - # - "k8s.statefulset.name" - # - "k8s.daemonset.name" - # - "k8s.cronjob.name" - # - "k8s.job.name" - # - "k8s.container.name" - # - "k8s.node.name" - # - "service.name" - # application_name: "{{.Values.global.defaultApplicationName }}" - # subsystem_name: "{{.Values.global.defaultSubsystemName }}" - processors: - k8s_attributes: - filter: - node_from_env_var: KUBE_NODE_NAME - extract: - metadata: - - "k8s.namespace.name" - - "k8s.deployment.name" - - "k8s.statefulset.name" - - "k8s.daemonset.name" - - "k8s.cronjob.name" - - "k8s.job.name" - - "k8s.pod.name" - - "k8s.node.name" - memory_limiter: null # Will get the k8s resource limits - resourcedetection/env: - detectors: ["system","env"] - timeout: 2s - override: false - # spanmetrics: - # metrics_exporter: coralogix - # dimensions: - # - name: "k8s.deployment.name" - # - name: "k8s.statefulset.name" - # - name: "k8s.daemonset.name" - # - name: "k8s.cronjob.name" - # - name: "k8s.job.name" - # - name: "k8s.container.name" - # - name: "k8s.node.name" - # - name: "k8s.namespace.name" - receivers: - otlp: - protocols: - grpc: - endpoint: ${MY_POD_IP}:4317 - http: - endpoint: ${MY_POD_IP}:4318 - zipkin: - endpoint: ${MY_POD_IP}:9411 - jaeger: - protocols: - grpc: - endpoint: ${MY_POD_IP}:14250 - thrift_http: - endpoint: ${MY_POD_IP}:14268 - thrift_compact: - endpoint: ${MY_POD_IP}:6831 - thrift_binary: - endpoint: ${MY_POD_IP}:6832 - prometheus: - config: - scrape_configs: - - job_name: opentelemetry-collector - scrape_interval: 30s - static_configs: - - targets: - - ${MY_POD_IP}:8888 - service: - extensions: - # - zpages - # - pprof - - health_check -# - memory_ballast - telemetry: - metrics: - address: ${env:MY_POD_IP}:8888 - pipelines: - traces: - exporters: - - loadbalancing - processors: - - memory_limiter - # - spanmetrics - - batch - receivers: - - otlp - metrics: null - logs: null -tolerations: - - operator: Exists - -resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 1 - memory: 2G - -ports: - jaeger-binary: - enabled: true - containerPort: 6832 - servicePort: 6832 - hostPort: 6832 - protocol: TCP - # In order to enable podMonitor, following part must be enabled in order to expose the required port: - # metrics: - # enabled: true - -# podMonitor: -# enabled: true - -# prometheusRule: -# enabled: true -# defaultRules: -# enabled: true - -# Annotations to be added to pod -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-demand-contour-ext - operator: NotIn - values: - - dedicated - - key: p-demand-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-demand-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-demand-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-demand-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert \ No newline at end of file diff --git a/helm-overrides/gke-demand-prd-ase1a/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index cc9afe8..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,133 +0,0 @@ -fullnameOverride: opentelemetry-demand-prd - -mode: daemonset - -priorityClassName: "system-node-critical" - -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.153.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-demand-contour-ext - operator: NotIn - values: - - dedicated - - key: p-demand-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-demand-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-demand-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-demand-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 100m - memory: 124Mi - limits: - cpu: 200m - memory: 256Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "demand" - team: "sre" - service: "opentelemetry-demand-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/gke-demand-prd-ase1a/prometheus-node-exporter/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index a250d02..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,487 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 200m - # memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-demand-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "demand" - team: "demand-sre" - service: "node-exporter-demand-a-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/gke-demand-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index a4b98a7..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,169 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-demand-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-demand-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "demand" - team: "demand-sre" - service: "stackdriver-exporter-demand-a-prd" - env: "prd" - priority: "p0" - type: "exporter" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-demand-prd-0622" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'compute.googleapis.com/instance,cloudsql.googleapis.com/database,kubernetes.io/node/ephemeral_storage/used_bytes,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - cloud.google.com/compute-class: demand-devops - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - effect: "NoSchedule" - key: "cloud.google.com/compute-class" - value: "demand-devops" - operator: "Equal" - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-dmnd-dmsre-stackdriver-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - } - - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/gke-demand-prd-ase1a/telegraf-operator/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/telegraf-operator/custom-values.yaml deleted file mode 100644 index 952023f..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,292 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - - infra-histogram-expiration-enabled: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - expiration_interval = "2m" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - - infra-telegraf-cardinality-optimized: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - expiration_interval = "2m" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "demand" - team: "demand-sre" - service: "telegraf-operator-demand-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - cloud.google.com/compute-class: "demand-devops" - -tolerations: - - effect: "NoSchedule" - key: "cloud.google.com/compute-class" - value: "demand-devops" - operator: "Equal" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false \ No newline at end of file diff --git a/helm-overrides/gke-demand-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml deleted file mode 100644 index 7fa3bbc..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml +++ /dev/null @@ -1,296 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-demand-prd-dr -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-demand-prd-dr-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent-dr - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dmnd-dmsre-vmagent-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - - https://vminsert-infra-prd-dr.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "demand" - team: "demand-sre" - service: "vmagent-demand-prd" - env: "prd" - priority: "p0" - type: "vmagent-dr" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-demand-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: metricsapi-demand-prd-dr.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 12 - memory: 10Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - cloud.google.com/compute-class: "vmagent-dr" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/gke-demand-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml deleted file mode 100644 index 2e1a48c..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml +++ /dev/null @@ -1,272 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 1 - -fullnameOverride: vmagent-demand-a-prd-fb -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-demand-prd-a-fb-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - -# Remove topology spread constraints -topologySpreadConstraints: [] - -# Modify deployment strategy -deployment: - enabled: true - -# Remove HPA -horizontalPodAutoscaler: - enabled: false - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-infr-sre-vmagent-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -# Remove pod disruption budget -podDisruptionBudget: - enabled: false - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-demand.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vmsinglenode-infra-prd.victoriametrics.svc.clusterset.local:8428/api/v1/write - - https://vmsinglenode-infra-prd.meeshogcp.in/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "demand-fb" - team: "sre" - service: "vmagent-demand-a-prd-fb" - env: "prd" - priority: "p0" - type: "vmagent-fb" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-demand-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-demand-a-prd-fb.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -# Modify resources -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - requests: - cpu: 2 - memory: 3Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - cloud.google.com/compute-class: "vmagent-c4d" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-c4d" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/gke-demand-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml deleted file mode 100644 index 6895a34..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert-dr - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-demand-prd-dr - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/demand/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-demand-prd-dr.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-demand-prd-dr" - env: "prd" - priority: "p0" - type: "vmalert-dr" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmagent-dr" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-demand-prd-ase1a/victoria-metrics-alert-secured-stateful/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/victoria-metrics-alert-secured-stateful/custom-values.yaml deleted file mode 100644 index cf33ad6..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/victoria-metrics-alert-secured-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-stateful-secured-demand-a-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-demand-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-demand-a-prd.victoriametrics.svc.cluster.local:8480/insert/200/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-demand-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/sensitive/demand/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: false - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-demand-a-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "demand" - team: "sre" - service: "vmalert-stateful-secured-demand-a-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-secured-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-demand-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml deleted file mode 100644 index ca333b0..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-demand-a-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-demand-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-demand-a-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-demand-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/demand/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-demand-a-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 1500m - memory: 3Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "demand" - team: "sre" - service: "vmalert-demand-a-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-demand-prd-ase1a/victoriametrics-agent/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index 08c9b23..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,488 +0,0 @@ -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "demand-a-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -# -- Replica count -replicaCount: 2 - -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - -# -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables -deployment: - spec: - minReadySeconds: 1200 - progressDeadlineSeconds: 1800 - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - -image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - # -- Image tag, set to `Chart.AppVersion` by default - tag: v1.133.0 # rewrites Chart.AppVersion - # -- Variant of the image to use. - # e.g. enterprise, scratch - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - -# -- Image pull secrets -imagePullSecrets: [] - -# -- Add additional DNS entries to pods hosts file. Check [official documentation](https://kubernetes.io/docs/tasks/network/customize-hosts-file-for-pods/) -hostAliases: [] -# - ip: 192.168.1.1 -# hostNames: -# - test.example.com -# - another.example.net - -# -- Override chart name -nameOverride: "" - -# -- Override resources fullname -fullnameOverride: "vm-agent-demand-a-prd" - -# -- Container working directory -containerWorkingDir: "/" - -rbac: - # -- Enables Role/RoleBinding creation - create: true - - # -- Role/RoleBinding annotations - annotations: {} - - # -- Role/RoleBinding labels - extraLabels: {} - - # -- If true and `rbac.enabled`, will deploy a Role/RoleBinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - - # -- additional rules for a role - extraRules: [] - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - # -- Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dmnd-sre-vmagnt-prd-mds@meesho-demand-prd-0622.iam.gserviceaccount.com - } - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # -- mount API token to pod directly - automountToken: true - -# -- See `kubectl explain poddisruptionbudget.spec` for more or check [official documentation](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# -- Generates `remoteWrite.*` flags and config maps with value content for values, that are of type list of map. -# Each item should contain `url` param to pass validation. -remoteWrite: - - url: http://vm-insert-demand-a-prd.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write - disableOnDiskQueue: false - dropSamplesOnOverload: false - # - url: http://vm-insert-demand-prd.victoriametrics.svc.clusterset.local:8480/insert/multitenant/prometheus/api/v1/write - # disableOnDiskQueue: false - # dropSamplesOnOverload: false - - url: http://prd-census-server-demand.prd-census-server-demand.svc.cluster.local/api/v1/write - disableOnDiskQueue: true - dropSamplesOnOverload: true -# urlRelabelConfig: -# - action: keep -# source_labels: [env] -# regex: "dev" - -# -- VMAgent extra command line arguments -extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8429 - promscrape.config.strictParse: false # default is true - promscrape.maxScrapeSize: 1000000000 # default is 167772160 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - # promscrape.suppressScrapeErrors: true # default is false - - # Uncomment and specify the port if you want to support any of the protocols: - # https://docs.victoriametrics.com/victoriametrics/vmagent/#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for more details. -env: - - name: GOGC - value: "200" - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth-secret - # key: password - -# -- Specify alternative source for env variables -envFrom: - [] - #- configMapRef: - # name: special-config - -# -- Extra labels for Deployment and Statefulset -extraLabels: - bu: "demand" - team: "demand-sre" - service: "vm-agent-demand-a-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra labels for Pods only -podLabels: - bu: "demand" - team: "demand-sre" - service: "vm-agent-demand-a-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra selector labels common for pod and service -selectorLabels: {} - -# -- Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# -- Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# -- Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -# -- Extra containers to run in a pod with vmagent -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -# -- Init containers for vmagent -initContainers: - [] - # - name: example - # image: example-image - -# -- Security context to be added to pod -podSecurityContext: - enabled: true - # fsGroup: 2000 - -# -- Security context to be added to pod's containers -securityContext: - enabled: true - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - # -- Enable agent service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - extraLabels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) for details - externalIPs: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8429 - # -- Target port - targetPort: http - # nodePort: 30000 - # -- Service type - type: ClusterIP - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - -ingress: - # -- Enable deployment of ingress for agent - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-agent-demand-a-prd.meeshogcp.in - path: - - / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - - # -- Ingress controller class name - ingressClassName: "nginx-internal" - - # -- Ingress path type - pathType: Prefix - -# -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) -resources: - requests: - cpu: 130 - memory: 250Gi - -# -- Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) -nodeSelector: - cloud.google.com/compute-class: "vmagent-c4d" - -# -- Node tolerations for server scheduling to nodes with taints. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-c4d" - effect: "NoSchedule" - -# -- Pod topologySpreadConstraints -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# -- Pod affinity -affinity: {} - -# -- VMAgent [scraping configuration](https://docs.victoriametrics.com/victoriametrics/vmagent/#how-to-collect-metrics-in-prometheus-format) -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-demand-a-prd" - -# -- Priority class to be assigned to the pod(s) -priorityClassName: "" - -# -- Enable the host network -hostNetwork: false - -serviceMonitor: - # -- Enable deployment of Service Monitor for server component. This is Prometheus operator object - enabled: false - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Service Monitor relabelings - relabelings: [] - # -- Basic auth params for Service Monitor - basicAuth: {} - # -- Service Monitor metricRelabelings - metricRelabelings: [] - # -- Service Monitor targetPort - targetPort: http - # interval: 15s - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - -# -- Empty dir configuration for a case, when persistence is disabled -emptyDir: {} - -persistentVolume: - # -- Create/use Persistent Volume Claim for server component. Empty dir if false - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Size of the volume. Should be calculated based on the logs you send and retention policy you set. - size: 10Gi - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume additional labels - extraLabels: {} - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Horizontal Pod Autoscaling. -# Note that it is not intended to be used for vmagents which perform scraping. -# In order to scale scraping vmagents check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) -horizontalPodAutoscaling: - # -- Use HPA for vmagent - enabled: false - # -- Maximum replicas for HPA to use to to scale vmagent - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale vmagent - minReplicas: 1 - # -- Metric for HPA to use to scale vmagent - metrics: [] - -# -- VMAgent scrape configuration -config: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -probe: - # -- Readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - # -- Liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - # -- Startup probe - startup: {} - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -allowedMetricsEndpoints: - - /metrics - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-demand-prd-ase1a/victoriametrics-insert/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/victoriametrics-insert/custom-values.yaml deleted file mode 100644 index 5eeecdf..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/victoriametrics-insert/custom-values.yaml +++ /dev/null @@ -1,411 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: false - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- IDs of vmstorage nodes to exclude from writing - excludeStorageIDs: [] - # -- Override default `app` label name - name: vminsert - # -- VMInsert strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - # override Chart.AppVersion - tag: v1.107.0-cluster - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMInsert http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-demand-a-prd - # -- Extra command line arguments for vminsert component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8480 - maxLabelsPerTimeseries: 55 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-demand-a-prd-0.vm-storage-demand-a-prd.victoriametrics.svc:8400" - - "vm-storage-demand-a-prd-1.vm-storage-demand-a-prd.victoriametrics.svc:8400" - - "vm-storage-demand-a-prd-2.vm-storage-demand-a-prd.victoriametrics.svc:8400" - - "vm-storage-demand-a-prd-3.vm-storage-demand-a-prd.victoriametrics.svc:8400" - - "vm-storage-demand-a-prd-4.vm-storage-demand-a-prd.victoriametrics.svc:8400" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "demand" - team: "demand-sre" - service: "vm-insert-demand-a-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Pod's additional labels - podLabels: - bu: "demand" - team: "demand-sre" - service: "vm-insert-demand-a-prd" - env: "prd" - priority: "p0" - type: "vminsert" - terminationGracePeriodSeconds: 30 - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - - # -- Readiness & Liveness probes - probe: - # -- VMInsert readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert startup probe - startup: {} - - # -- Relabel configuration - relabel: - enabled: false - config: [] - # -- Use existing configmap if specified - # otherwise .config values will be used. Relabel config **should** reside under `relabel.yml` key - configMap: "" - - # Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 7 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vminsert - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vminsert - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - cloud.google.com/compute-class: "vminsert-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 20 - memory: 10Gi - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - service: - # -- Create VMInsert service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here]( https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) for details - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-insert-demand-a-prd.meeshogcp.in - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - - # -- Ingress controller class name - ingressClassName: nginx-internal - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for insert component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/insert" .Values.vminsert }}' - - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vminsert component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-demand-prd-ase1a/victoriametrics-select/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/victoriametrics-select/custom-values.yaml deleted file mode 100644 index 19c47cb..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/victoriametrics-select/custom-values.yaml +++ /dev/null @@ -1,449 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - splitService: true - externalService: - enabled: true - name: vmselect-demand-a-prd-proxy - - # -- Enable clusternative service for LoadBalancer vmselect component - clusternativeService: - enabled: false - # -- Override default `app` label name - name: "" - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # override Chart.AppVersion - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMSelect http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-demand-a-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - # -- Extra command line arguments for vmselect component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8481 - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-demand-a-prd-0.vm-storage-demand-a-prd.victoriametrics.svc:8401" - - "vm-storage-demand-a-prd-1.vm-storage-demand-a-prd.victoriametrics.svc:8401" - - "vm-storage-demand-a-prd-2.vm-storage-demand-a-prd.victoriametrics.svc:8401" - - "vm-storage-demand-a-prd-3.vm-storage-demand-a-prd.victoriametrics.svc:8401" - - "vm-storage-demand-a-prd-4.vm-storage-demand-a-prd.victoriametrics.svc:8401" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "demand" - team: "demand-sre" - service: "vm-select-demand-a-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # -- Pod's additional labels - podLabels: - bu: "demand" - team: "demand-sre" - service: "vm-select-demand-a-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Readiness & Liveness probes - probe: - # -- VMSelect readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 20 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 7 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vmselect - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vmselect - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 36 - memory: 60Gi - - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Create VMSelect service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-select-demand-a.prd.meesho.int - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # -- Ingress controller class name - ingressClassName: contour-internal-0 - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for select component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/select" .Values.vmselect }}' - - # -- vmselect mode: deployment, daemonSet - mode: deployment - - # -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables - deployment: - spec: - # -- VMSelect strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - - # -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables - statefulSet: - enabled: false - spec: - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume extra labels - extraLabels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmselect component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: false - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - diff --git a/helm-overrides/gke-demand-prd-ase1a/victoriametrics-storage/custom-values.yaml b/helm-overrides/gke-demand-prd-ase1a/victoriametrics-storage/custom-values.yaml deleted file mode 100644 index 7452a58..0000000 --- a/helm-overrides/gke-demand-prd-ase1a/victoriametrics-storage/custom-values.yaml +++ /dev/null @@ -1,363 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - enabled: false - -vminsert: - enabled: false - -vmauth: - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- Override default `app` label name - name: vmstorage - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Variant of the image to use. e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMStorage http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-demand-a-prd - - # -- Additional environment variables (ex.: secret tokens, flags). Check https://docs.victoriametrics.com/victoriametrics/#environment-variables for details - env: [] - # -- Specify alternative source for env variables - envFrom: [] - # -- Data retention period. Possible units character: h(ours), d(ays), w(eeks), y(ears), if no unit character specified - month. The minimum retention period is 24h. - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - httpListenAddr: :8482 - loggerTimezone: "Asia/Kolkata" - dedup.minScrapeInterval: 60s - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmstorage-n4d" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - cloud.google.com/compute-class: "vmstorage-n4d" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - # -- Use an alternate scheduler, e.g. "stork". Check https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ for details - # schedulerName: - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Override Persistent Volume Claim name - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are https://kubernetes.io/docs/concepts/storage/persistent-volumes/ - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume extra labels - extraLabels: {} - # -- Storage class name. Will be empty if not set - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume - size: 9677Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "demand" - team: "demand-sre" - service: "vm-storage-demand-a-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - # -- Pod's additional labels - podLabels: - bu: "demand" - team: "demand-sre" - service: "vm-storage-demand-a-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Details are https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 85 - memory: 700Gi - - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: - enabled: false - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - podSecurityContext: - enabled: false - service: - enabled: true - # -- Service annotations - annotations: {} - # -- Service ClusterIP - clusterIP: None - # -- Service type - type: ClusterIP - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmselect - vmselectPort: 8401 - # -- Extra service ports - extraPorts: [] - # -- Health check node port for a service. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilyPolicy: "" - # -- List of service IP families. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilies: [] - # -- Traffic Distribution. Check https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution - trafficDistribution: "" - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - minReadySeconds: 5 - # -- Readiness probes - probe: - # -- VMStorage readiness probe - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMStorage startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmstorage component - enabled: false - # -- Maximum replicas for HPA to use to to scale the vmstorage component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmstorage component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmstorage component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: - scaleDown: - selectPolicy: Disabled - - vmbackupmanager: - # -- Enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/README.md b/helm-overrides/gke-dsgpu-prd-ase1a/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/aurva-dataplane/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/aurva-dataplane/custom-values.yaml deleted file mode 100644 index 6d8d167..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/aurva-dataplane/custom-values.yaml +++ /dev/null @@ -1,661 +0,0 @@ -## Aurva Data Plane -## Ref: https://github.com/aurva-io/aurva-charts.git - -postgresql: - enabled: true - fullnameOverride: "aurva-dataplane-database" - volumePermissions: - ## @param volumePermissions.enabled Enable init container that changes the owner and group of the persistent volume - ## - enabled: true - global: - storageClass: pd-standard-retain - postgresql: - auth: - postgresPassword: "aurva" - database: "controller" - # Add toleration to make sure where this postgres db pod should reside (Applicable for production workloads): For more detail ref: https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/ - primary: - extendedConfiguration: | - max_connections = 300 - - # #PLACEHOLDER## - # tolerations: - # - effect: NoSchedule - # key: cloud.google.com/compute-class - # operator: Equal - # value: datascience-devops - - # # -- Select nodes to deploy which matches the following labels - # nodeSelector: ##PLACEHOLDER## - # dedicated: datascience-devops - -# -- Provide a name in place of `aurva` -# namespaceOverride: aurva-dataplane - -########################################################## -# Global Configs -########################################################## -global: - aurva_controller: - enabled: true - aurva_fastdet: - enabled: false - aurva_pii_analyzer: - enabled: true - aurva_ocr: - enabled: true - aurva_collector: - enabled: true - - deploymentAnnotations: {} - - priorityClassName: "" - -########################################################## -# Aurva Controller -########################################################## -aurva_controller: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "dsgpu" - team: "dsgpu-devops" - service: "aurva-dsgpu-prd" - env: "prd" - priority: "p0" - type: "aurva_controller" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 10 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - image: - # Image of the app container - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-controller/aurva-controller - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - - type: secret - name: aurva-controller-secrets - # - type: configmap - # name: proxy-datasource-config - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 2 - requests: - memory: 1Gi - cpu: 1 - - aurvaFastdet: - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-fastdet/aurva-fastdet - tag: "v2.30.13" - pullPolicy: IfNotPresent - envFrom: [] - env: [] - resources: - limits: - cpu: 0.5 - memory: 512Mi - requests: - cpu: 0.5 - memory: 512Mi - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: datascience-devops - - # -- Taint tolerations for nodes - ##PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-controller-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - config: - #variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - COMMAND_URL: "command.aurva-prd.meeshogcp.in:80" - DEPLOYMENT_TYPE: "kubernetes" - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - FLUSHER_WORKER_POOL_SIZE: "1000" - FLUSHER_BATCH_SIZE: "30000" - UNIQUENESS_IDENTIFIER: "k8s-dsgpu-prd-ase1" #Recommendation: should be equal to cluster name - PROVIDER_ACCOUNT_ID: "meesho-datascience-prd-0622" # GCP PROJECT ID (not Number) - REGION: "asia-southeast1" #eg: asia-south1 - ENVIRONMENT: "prod" - OCR_ENABLED: "true" - MONITORING_ENABLED: "false" - HYBRID_ONLY_MODE: "false" - FORCE_TLS: "false" - FASTDET_FLAG : "true" - AADHAAR_ENHANCER: "0" - WORKSPACE_EVENT_TRACKING_ENABLED: "false" - ACCESS_IQ_ENABLED: "true" - GRPC_ENFORCE_ALPN_ENABLED: "false" - ENABLE_FASTDET: "true" - GOGC: "70" - #pii data - PII_BUCKET_NAME: "gcs-infra-devop-aurva-dsgpu-prd" - PII_LOG_BUCKET_REGION: "asia-southeast1" - PII_LOG_CRON: "*/5 * * * *" - ENABLE_PII_LOG: "true" - PII_EVIDENCE_UPLOAD_MAX_BLOCKING_TASKS: "50000" - PII_EVIDENCE_UPLOAD_MAX_CONCURRENT_TASKS: "500" - PII_EVIDENCE_MAX_CACHE_WEIGHT: "100" - QUOTA_CLEANUP_CRON: "0 0 * * *" - ENABLE_PII_QUOTA: "true" - MAX_PII_EVIDENCES_PER_KEY_PER_WINDOW: "3" - PII_QUOTA_SYNC_CRON: "*/2 * * * *" - QUOTA_WINDOW_HOURS: "24" - #constants - SKIP_NAMESPACES: "contour-internal-1-dsgpu-prd,contour-internal-0-dsgpu-prd,contour-external-dsgpu-prd" - CLOUD_PROVIDER: "gcp" - LOG_ENV: "production" - RDS_SCANNER_AVAILABILITY : "false" - REDSHIFT_SCANNER_AVAILABILITY : "false" - S3_SCANNER_AVAILABILITY : "false" - DYNAMO_SCANNER_AVAILABILITY: "false" - DOCDB_SCANNER_AVAILABILITY: "false" - OPENSEARCH_SCANNER_AVAILABILITY: "false" - CLOUDSQL_SCANNER_AVAILABILITY: "true" - BIGQUERY_SCANNER_AVAILABILITY: "true" - AWS_SNAPSHOT_SCANNER_AVAILABILITY: "false" - CLOUDSTORAGE_SCANNER_AVAILABILITY: "true" - KEYSPACES_SCANNER_AVAILABILITY: "false" - ALLOYDB_SCANNER_AVAILABILITY: "true" - BIGTABLE_SCANNER_AVAILABILITY: "true" - GCP_BACKUP_AVAILABILITY: "true" - EGRESS_MODE_ONLY: "false" - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - SCAN_UUID_ENABLED: "true" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-controller-sa - # -- Annotations applied to created service account - annotations: - iam.gke.io/gcp-service-account: sa-datascience-prd-aurva-contr@meesho-datascience-prd-0622.iam.gserviceaccount.com -# eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 15 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 - - -########################################################## -# Aurva OCR -########################################################## -aurva_ocr: - # -- Additional labels for aurva-controller - additionalLabels: - bu: "dsgpu" - team: "dsgpu-devops" - service: "aurva-dsgpu-prd" - env: "prd" - priority: "p0" - type: "aurva_ocr" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 1 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-ocr/aurva-ocr - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-ocr: - type: secret - name: aurva-ocr-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 1 - requests: - memory: 2Gi - cpu: 1 - - # -- Select nodes to deploy which matches the following labels - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: datascience-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-ocr-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - OCR_TIME_LIMIT: "1" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-ocr-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # iam.gke.io/gcp-service-account: service-account@gcp.iam.gserviceaccount.com - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva Collector -########################################################## -aurva_collector: - - # -- Additional labels for aurva-analyzer - additionalLabels: - bu: "dsgpu" - team: "dsgpu-devops" - service: "aurva-dsgpu-prd" - env: "prd" - priority: "p0" - type: "aurva_collector" - - - # -- Annotations on aurva-analyzer - annotations: {} -# "key": "value" - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-collector/aurva-collector - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-controller: - type: secret - name: aurva-collector-secrets - - resources: - limits: - cpu: 800m - memory: 800Mi - requests: - cpu: 200m - memory: 512Mi - - podSecurityContext: {} - - securityContext: - privileged: false - capabilities: - add: - # For kernel v5.8 and above we don't need CAP_SYS_ADMIN or CAP_SYS_RESOURCE - # we just need CAP_BPF and CAP_PERFMON. This has been tested on our EKS node - # which is on kernel v5.10.x - # When SSL Tracing is required we need CAP_SYS_ADMIN and CAP_SYS_PTRACE - # on top of the previous capabilities - # So finally these are the 4 possible combinations for capabilies - # 1. Newer Kernels without SSL - - BPF - - PERFMON - # 2. Newer Kernels with SSL - # - SYS_ADMIN - # - SYS_PTRACE - # 3. Older Kernels without SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # 4. Older Kernels with SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # - SYS_PTRACE - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - volumes: - - name: debugfs - mountPath: /sys/kernel/debug - hostPath: /sys/kernel/debug - - name: vmlinux - mountPath: /sys/kernel/btf/vmlinux - hostPath: /sys/kernel/btf/vmlinux - - name: procfs - mountPath: /host/proc - hostPath: /proc - - # -- Taint tolerations for nodes - tolerations: - # - effect: NoSchedule - # key: cloud.google.com/compute-class - # operator: Equal - # value: megatetra - - operator: Exists - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - c3-highcpu-44-vmselect-compute-class - - n4-highcpu-16-vminsert-compute-class - - n4d-highcpu-64-vmagent-compute-class - - n4d-highmem-48-vmstorage-compute-class - - c3-highcpu-22-contour-internal-0-compute-class - - c3-highcpu-22-contour-internal-1-compute-class - - c4d-highcpu-16-contour-internal-0-compute-class - - c4d-highcpu-16-contour-internal-1-compute-class - - alloy - - preprod-spot-16 - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-collector-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - # variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - TRACE_INTERNAL_SVC: "true" - TRACE_INTERNAL_SVC_HTTP: "true" - INTERNAL_SVC_SAMPLE_INTERVAL: "2m" - TRACE_SSL: "false" - TRACE_PSQL: "false" - TRACE_SQLSERVER: "false" - TRACE_MYSQL: "false" - TRACE_EGRESS: "true" - TRACE_GO_TLS: "false" - TRACE_ML_SERVICES: "false" - # constants - LOG_ENV: production - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - MONITORING_ENABLED: "false" - ENABLE_INGRESS_INFORMER: "false" - ENABLE_SERVICE_INFORMER: "false" - ENABLE_ISTIO_INFORMER: "false" - EXCLUDED_PII_REGEX_TYPES: "ip_address,us_bank_number,us_driver_license,us_itin,us_passport,us_routing,us_mbi,ssn" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-collector-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva PII Analyzer -########################################################## -aurva_pii_analyzer: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "dsgpu" - team: "dsgpu-devops" - service: "aurva-dsgpu-prd" - env: "prd" - priority: "p0" - type: "aurva_pii_analyzer" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 3 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - ##PLACEHOLDER## - nodeSelector: - cloud.google.com/compute-class: datascience-devops - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-piianalyzer/aurva-piianalyzer - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-pii-analyzer: - type: secret - name: aurva-pii-analyzer-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 4Gi - cpu: 4 - requests: - memory: 2Gi - cpu: 2 - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: datascience-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-pii-analyzer-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - SCHEDULER_TIME: "1" - SUPPORTED_REGION: "US" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-pii-analyzer-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 3 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/cert-manager/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/cert-manager/custom-values.yaml deleted file mode 100644 index 34c382f..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/cert-manager/custom-values.yaml +++ /dev/null @@ -1,129 +0,0 @@ -global: - logLevel: 2 - rbac: - create: true - priorityClassName: "high-priority" -installCRDs: false - -crds: - enabled: true - keep: true - -# Cert-manager Controller -replicaCount: 1 -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-controller - tag: v1.20.1 - pullPolicy: IfNotPresent - -nodeSelector: - cloud.google.com/compute-class: datascience-devops - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - -resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 500m - memory: 512Mi - -# Webhook Configuration -webhook: - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-webhook - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: datascience-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# CA Injector Configuration -cainjector: - enabled: true - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-cainjector - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: datascience-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# ACME Solver Configuration -acmesolver: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-acmesolver - tag: v1.20.1 - pullPolicy: IfNotPresent - -# Startup API Check -startupapicheck: - enabled: true - timeout: 1m - backoffLimit: 4 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-startupapicheck - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: datascience-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - -# Prometheus Monitoring -prometheus: - enabled: true - servicemonitor: - enabled: false - interval: 60s - scrapeTimeout: 30s - labels: - prometheus: cert-manager diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/a2-highgpu-1-imgcache-compute-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/a2-highgpu-1-imgcache-compute-class.yaml deleted file mode 100644 index 800958c..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/a2-highgpu-1-imgcache-compute-class.yaml +++ /dev/null @@ -1,52 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: a2-highgpu-1-imgcache-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: a2-highgpu-1-imgcache-compute-class - autoRepair: true - autoUpgrade: false - imageStreaming: - enabled: true - imageType: cos_containerd - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: latest - type: nvidia-tesla-a100 - location: - zones: - - asia-southeast1-b - machineType: a2-highgpu-1g - maxPodsPerNode: 20 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - secondaryBootDisks: - - diskImageName: image-dsci-mlp-imagecache-v1-prd-ase1 - mode: CONTAINER_IMAGE_CACHE - - gpu: - count: 1 - driverVersion: latest - type: nvidia-tesla-a100 - location: - zones: - - asia-southeast1-c - machineType: a2-highgpu-1g - maxPodsPerNode: 20 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - secondaryBootDisks: - - diskImageName: projects/meesho-datascience-prd-0622/global/images/image-dsci-mlp-imagecache-v1-prd-ase1 - mode: CONTAINER_IMAGE_CACHE - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3-highcpu-22-contour-internal-0-compute-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3-highcpu-22-contour-internal-0-compute-class.yaml deleted file mode 100644 index 923ca25..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3-highcpu-22-contour-internal-0-compute-class.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: c3-highcpu-22-contour-internal-0-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: c3-highcpu-22-contour-internal-0-compute-class - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3-highcpu-22 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3-highcpu-22-contour-internal-1-compute-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3-highcpu-22-contour-internal-1-compute-class.yaml deleted file mode 100644 index 7567827..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3-highcpu-22-contour-internal-1-compute-class.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: c3-highcpu-22-contour-internal-1-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: c3-highcpu-22-contour-internal-1-compute-class - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3-highcpu-22 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3-highcpu-4-nginx-internal-compute-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3-highcpu-4-nginx-internal-compute-class.yaml deleted file mode 100644 index 15b83a5..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3-highcpu-4-nginx-internal-compute-class.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: c3-highcpu-4-nginx-internal-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: c3-highcpu-4-nginx-internal-compute-class - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3-highcpu-4 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 50 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3-highcpu-44-vmselect-compute-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3-highcpu-44-vmselect-compute-class.yaml deleted file mode 100644 index 1a06c34..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3-highcpu-44-vmselect-compute-class.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: c3-highcpu-44-vmselect-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: c3-highcpu-44-vmselect-compute-class - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3-highcpu-44 - maxPodsPerNode: 22 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3d-highcpu-8-contour-internal-0-compute-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3d-highcpu-8-contour-internal-0-compute-class.yaml deleted file mode 100644 index 78c9548..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3d-highcpu-8-contour-internal-0-compute-class.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: c3d-highcpu-8-contour-internal-0-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: c3d-highcpu-8-contour-internal-0-compute-class - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3d-highcpu-8-contour-internal-1-compute-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3d-highcpu-8-contour-internal-1-compute-class.yaml deleted file mode 100644 index f6468b5..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c3d-highcpu-8-contour-internal-1-compute-class.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: c3d-highcpu-8-contour-internal-1-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: c3d-highcpu-8-contour-internal-1-compute-class - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c3d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c4d-highcpu-16-contour-internal-0-compute-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c4d-highcpu-16-contour-internal-0-compute-class.yaml deleted file mode 100644 index ee4504b..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c4d-highcpu-16-contour-internal-0-compute-class.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: c4d-highcpu-16-contour-internal-0-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: c4d-highcpu-16-contour-internal-0-compute-class - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c4d-highcpu-16-contour-internal-1-compute-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c4d-highcpu-16-contour-internal-1-compute-class.yaml deleted file mode 100644 index 98d5cea..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/c4d-highcpu-16-contour-internal-1-compute-class.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: c4d-highcpu-16-contour-internal-1-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: c4d-highcpu-16-contour-internal-1-compute-class - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - location: - zones: - - asia-southeast1-a - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/datascience-devops.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/datascience-devops.yaml deleted file mode 100644 index 8495f24..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/datascience-devops.yaml +++ /dev/null @@ -1,36 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: datascience-devops -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: datascience-devops - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - priorities: - - machineType: n4-standard-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 60 - bootDiskType: hyperdisk-balanced - - machineType: c4-standard-4 - spot: false - maxPodsPerNode: 32 - storage: - bootDiskSize: 60 - bootDiskType: hyperdisk-balanced - - machineType: e2-standard-4 - spot: false - maxPodsPerNode: 32 - storage: - bootDiskSize: 60 - bootDiskType: pd-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/dsgpu-kyverno-cc.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/dsgpu-kyverno-cc.yaml deleted file mode 100644 index 4118879..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/dsgpu-kyverno-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dsgpu-kyverno -spec: - nodePoolConfig: - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: dsgpu-kyverno - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/e2-standard-4-datascience-devops-compute-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/e2-standard-4-datascience-devops-compute-class.yaml deleted file mode 100644 index 29e4642..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/e2-standard-4-datascience-devops-compute-class.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: e2-standard-4-datascience-devops-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: e2-standard-4-datascience-devops-compute-class - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - location: - zones: - - asia-southeast1-a - machineType: e2-standard-4 - maxPodsPerNode: 64 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-l4-priority-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-l4-priority-class.yaml deleted file mode 100644 index 7ef17d3..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-l4-priority-class.yaml +++ /dev/null @@ -1,51 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-l4-priority-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: g2-l4-priority-class - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-a - machineType: g2-standard-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-b - machineType: g2-standard-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-c - machineType: g2-standard-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-16-l4-compute-class-ld.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-16-l4-compute-class-ld.yaml deleted file mode 100644 index 963f4ca..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-16-l4-compute-class-ld.yaml +++ /dev/null @@ -1,51 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-16-l4-compute-class-ld -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: g2-standard-16-l4-compute-class-ld - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-a - machineType: g2-standard-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-b - machineType: g2-standard-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-c - machineType: g2-standard-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-16-l4-compute-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-16-l4-compute-class.yaml deleted file mode 100644 index 464b144..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-16-l4-compute-class.yaml +++ /dev/null @@ -1,51 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-16-l4-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: g2-standard-16-l4-compute-class - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-a - machineType: g2-standard-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-b - machineType: g2-standard-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-c - machineType: g2-standard-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-4-l4-compute-class-driver-latest.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-4-l4-compute-class-driver-latest.yaml deleted file mode 100644 index acedfb3..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-4-l4-compute-class-driver-latest.yaml +++ /dev/null @@ -1,51 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-4-l4-compute-class-driver-latest -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: g2-standard-4-l4-compute-class-driver-latest - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-a - machineType: g2-standard-4 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-b - machineType: g2-standard-4 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-c - machineType: g2-standard-4 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-4-l4-compute-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-4-l4-compute-class.yaml deleted file mode 100644 index 3329c90..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-4-l4-compute-class.yaml +++ /dev/null @@ -1,51 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-4-l4-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: g2-standard-4-l4-compute-class - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-a - machineType: g2-standard-4 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-b - machineType: g2-standard-4 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-c - machineType: g2-standard-4 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-8-l4-300gb-compute-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-8-l4-300gb-compute-class.yaml deleted file mode 100644 index 6cb9d4e..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-8-l4-300gb-compute-class.yaml +++ /dev/null @@ -1,54 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-8-l4-300gb-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: g2-standard-8-l4-300gb-compute-class - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-a - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskSize: 300 - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-b - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskSize: 300 - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-c - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskSize: 300 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-8-l4-compute-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-8-l4-compute-class.yaml deleted file mode 100644 index 242637a..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-standard-8-l4-compute-class.yaml +++ /dev/null @@ -1,51 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-8-l4-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: g2-standard-8-l4-compute-class - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-a - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-b - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-c - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-std-16-l4-priority-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-std-16-l4-priority-class.yaml deleted file mode 100644 index 097bddc..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/g2-std-16-l4-priority-class.yaml +++ /dev/null @@ -1,51 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-std-16-l4-priority-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: g2-std-16-l4-priority-class - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-a - machineType: g2-standard-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-b - machineType: g2-standard-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: latest - type: nvidia-l4 - location: - zones: - - asia-southeast1-c - machineType: g2-standard-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/n4-highcpu-16-vminsert-compute-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/n4-highcpu-16-vminsert-compute-class.yaml deleted file mode 100644 index 9193d1a..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/n4-highcpu-16-vminsert-compute-class.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: n4-highcpu-16-vminsert-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: n4-highcpu-16-vminsert-compute-class - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4-highcpu-16 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/n4d-highcpu-64-vmagent-compute-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/n4d-highcpu-64-vmagent-compute-class.yaml deleted file mode 100644 index c8905c0..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/n4d-highcpu-64-vmagent-compute-class.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: n4d-highcpu-64-vmagent-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: n4d-highcpu-64-vmagent-compute-class - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4d-highcpu-64 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/n4d-highmem-48-vmstorage-compute-class.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/n4d-highmem-48-vmstorage-compute-class.yaml deleted file mode 100644 index 2e42b0b..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/computeclass/n4d-highmem-48-vmstorage-compute-class.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: n4d-highmem-48-vmstorage-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: n4d-highmem-48-vmstorage-compute-class - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - location: - zones: - - asia-southeast1-a - machineType: n4d-highmem-48 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/conntrack-adjuster/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 993f9c4..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - # Maximum number of conntrack entries - max: 2097152 - # Hash size for conntrack - hashsize: 524288 - # Sleep interval between adjustments (seconds) - sleepInterval: 30 - -# Enable additional matchExpressions -addExtraMatchExpressions: true - -# Additional matchExpressions to append -additionalMatchExpressions: - - key: dedicated - operator: In - values: - - "megaquad" -# - "ingress" -# - key: environment -# operator: In -# values: -# - "production" -# - "staging" diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/contour-ca-issuer/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/contour-ca-issuer/custom-values.yaml deleted file mode 100644 index 73370d1..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/contour-ca-issuer/custom-values.yaml +++ /dev/null @@ -1,17 +0,0 @@ -# ClusterIssuer override for k8s-dsgpu-prd-ase1 (prd dsgpu cluster, ase1a zone). -# ClusterIssuer is a cluster-scoped resource — this file pins the issuer -# identity for this cluster so the name is auditable per-cluster. -# -# Must match the `issuerRef.name` in consuming Certificate CRs (see -# devops-helm-charts/2.0.0/templates/proxyless-grpc-cert.yaml). - -issuerName: contour-dsgpu-prd-ca-issuer -rootCASecretName: contour-dsgpu-ca - -externalSecret: - enabled: true - vaultPath: meesho/devops/contour/root-ca - refreshInterval: "0" - secretStoreRef: - name: vault-backend -namespace: cert-manager-dsgpu-prd \ No newline at end of file diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/contour-cert-checker/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/contour-cert-checker/custom-values.yaml deleted file mode 100644 index d7ec36a..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=gke-dsgpu-prd-ase1a"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: e2-standard-4-datascience-devops-compute-class - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/contour-internal-0/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/contour-internal-0/custom-values.yaml deleted file mode 100644 index 1cbcf8f..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,103 +0,0 @@ -fullnameOverride: "contour-internal-0-dsgpu-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 60s - max-connection-duration: 600s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: dsgpu - team: dsgpu-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 500m - memory: 3Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: c3d-highcpu-8-contour-internal-0-compute-class - nodeSelector: - cloud.google.com/compute-class: c3d-highcpu-8-contour-internal-0-compute-class - service: - tcpLB: true - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: dsgpu - team: dsgpu-devops - env: prd - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 5% - maxUnavailable: 0 - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: c3d-highcpu-8-contour-internal-0-compute-class - nodeSelector: - cloud.google.com/compute-class: c3d-highcpu-8-contour-internal-0-compute-class - logLevel: error - extraArgs: - - '--concurrency 6' - autoscaling: - enabled: true - minReplicas: 9 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - limits: - cpu: 6 - memory: 12Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-dsgpu-a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/contour-internal-1/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/contour-internal-1/custom-values.yaml deleted file mode 100644 index 746d7cb..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,118 +0,0 @@ -fullnameOverride: "contour-internal-1-dsgpu-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 400s - max-connection-duration: 540s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: dsgpu - team: dsgpu-devops - env: prd - manageCRDs: false - resources: - requests: - cpu: 6 - memory: 3Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: c3d-highcpu-8-contour-internal-1-compute-class - nodeSelector: - cloud.google.com/compute-class: c3d-highcpu-8-contour-internal-1-compute-class - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: dsgpu - team: dsgpu-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: c3d-highcpu-8-contour-internal-1-compute-class - nodeSelector: - cloud.google.com/compute-class: c3d-highcpu-8-contour-internal-1-compute-class - logLevel: error - terminationGracePeriodSeconds: 500 - extraArgs: - - '--concurrency 6' - autoscaling: - behaviour: - scaledown: - policies: - - periodseconds: 500 - type: Pods - value: 1 - selectpolicy: Min - stabilizationWindowSeconds: 600 - scaleup: - policies: - - periodseconds: 15 - type: Pods - value: 2 - selectpolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "35" - targetMemory: "40" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - limits: - cpu: 6 - memory: 12Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int1-dsgpu-a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/contour-internal-intra-0/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index e6ae6c1..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,101 +0,0 @@ -fullnameOverride: "contour-internal-0-dsgpu-intra-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 60s - max-connection-duration: 600s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: dsgpu - team: dsgpu-devops - env: prd - manageCRDs: false - resources: - requests: - cpu: 500m - memory: 3Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: c3d-highcpu-8-contour-internal-0-compute-class - nodeSelector: - cloud.google.com/compute-class: c3d-highcpu-8-contour-internal-0-compute-class - service: - tcpLB: true - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: dsgpu - team: dsgpu-devops - env: prd - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 5% - maxUnavailable: 0 - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: c3d-highcpu-8-contour-internal-0-compute-class - nodeSelector: - cloud.google.com/compute-class: c3d-highcpu-8-contour-internal-0-compute-class - logLevel: error - extraArgs: - - '--concurrency 6' - autoscaling: - enabled: true - minReplicas: 9 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - limits: - cpu: 6 - memory: 12Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/contour-internal-intra-1/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index 82f7f5c..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -fullnameOverride: "contour-internal-1-dsgpu-intra-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 400s - max-connection-duration: 540s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: dsgpu - team: dsgpu-devops - env: prd - manageCRDs: false - resources: - requests: - cpu: 500m - memory: 3Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: c3d-highcpu-8-contour-internal-1-compute-class - nodeSelector: - cloud.google.com/compute-class: c3d-highcpu-8-contour-internal-1-compute-class - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: dsgpu - team: dsgpu-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: c3d-highcpu-8-contour-internal-1-compute-class - nodeSelector: - cloud.google.com/compute-class: c3d-highcpu-8-contour-internal-1-compute-class - logLevel: error - extraArgs: - - '--concurrency 6' - terminationGracePeriodSeconds: 500 - autoscaling: - behaviour: - scaledown: - policies: - - periodseconds: 500 - type: Pods - value: 1 - selectpolicy: Min - stabilizationWindowSeconds: 600 - scaleup: - policies: - - periodseconds: 15 - type: Pods - value: 2 - selectpolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 10 - maxReplicas: 250 - targetCPU: "35" - targetMemory: "40" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - limits: - cpu: 6 - memory: 12Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/coredns/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/coredns/custom-values.yaml deleted file mode 100644 index 8ba24c4..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/coredns/custom-values.yaml +++ /dev/null @@ -1,27 +0,0 @@ -replicaCount: 30 -communicationType: "intra" - -labels: - bu: dsgpu - team: datascience-devops - env: prd - -clusterIP: 10.1.64.2 - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - -nodeSelector: - cloud.google.com/compute-class: datascience-devops - kubernetes.io/os: linux diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/external-dns-services/contour-internal-0-dsgpu-prd-envoy-headless-external-dns-svc.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/external-dns-services/contour-internal-0-dsgpu-prd-envoy-headless-external-dns-svc.yaml deleted file mode 100644 index ec30ca6..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/external-dns-services/contour-internal-0-dsgpu-prd-envoy-headless-external-dns-svc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: v1 -kind: Service -metadata: - annotations: - external-dns.alpha.kubernetes.io/hostname: contour-internal-0-dsgpu-a.dsgpu.meesho.int - labels: - app.kubernetes.io/component: envoy - app.kubernetes.io/instance: contour-internal-0-dsgpu-prd - app.kubernetes.io/managed-by: Helm - app.kubernetes.io/name: contour - app.kubernetes.io/version: 1.27.0 - argocd.argoproj.io/instance: contour-internal-0-dsgpu-prd - helm.sh/chart: contour-15.0.0 - name: contour-internal-0-dsgpu-prd-envoy-headless-external-dns-svc - namespace: contour-internal-0-dsgpu-prd -spec: - clusterIP: None - clusterIPs: - - None - ipFamilies: - - IPv4 - ipFamilyPolicy: SingleStack - ports: - - name: http - port: 80 - targetPort: http - selector: - app.kubernetes.io/component: envoy - app.kubernetes.io/instance: contour-internal-0-dsgpu-prd - app.kubernetes.io/name: contour diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/external-dns/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/external-dns/custom-values.yaml deleted file mode 100644 index 125029e..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/external-dns/custom-values.yaml +++ /dev/null @@ -1,60 +0,0 @@ -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/external-dns - tag: "v0.13.4" - -fullnameOverride: prd-external-dns - -serviceAccount: - create: true - name: prd-external-dns - annotations: - iam.gke.io/gcp-service-account: sa-admin-external-dns-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - -provider: - name: google - -sources: - - service - -policy: upsert-only - -registry: txt -txtOwnerId: gke-externaldns-dsgpu-prd - -domainFilters: - - dsgpu.meesho.int - -interval: 10s - -logLevel: info -logFormat: text - -revisionHistoryLimit: 10 - -deploymentStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 0 - -priorityClassName: "high-priority" - -terminationGracePeriodSeconds: 30 -dnsPolicy: ClusterFirst - -podAnnotations: - cluster-autoscaler.kubernetes.io/safe-to-evict: "false" - -podLabels: - app: external-dns - -extraArgs: - google-batch-change-size: "1000" - google-project: meesho-admin-prd-0622 - -rbac: - create: true - additionalPermissions: - - apiGroups: [""] - resources: ["endpoints"] - verbs: ["get", "watch", "list"] diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/external-secrets/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/external-secrets/custom-values.yaml deleted file mode 100644 index 29867ec..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/external-secrets/custom-values.yaml +++ /dev/null @@ -1,61 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Annotations to add to Pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - nodeSelector: - cloud.google.com/compute-class: datascience-devops - certController: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - nodeSelector: - cloud.google.com/compute-class: datascience-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - webhook: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - nodeSelector: - cloud.google.com/compute-class: datascience-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" \ No newline at end of file diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/flagger/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/flagger/custom-values.yaml deleted file mode 100644 index 8be0037..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/flagger/custom-values.yaml +++ /dev/null @@ -1,62 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - cloud.google.com/compute-class: datascience-devops - -tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: datascience-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: dsgpu-devops - bu: infra - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-dsgpu-rollout-service.prd-dsgpu-rollout-service.svc.cluster.local" - -rolloutService: - enabled: true - apps: diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/fluentd/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/fluentd/custom-values.yaml deleted file mode 100644 index d8421b6..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/fluentd/custom-values.yaml +++ /dev/null @@ -1,731 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-dssre-fluentd-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: dsgpu - team: sre - type: fluentd - service: fluentd-dsgpu-a-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: dsgpu - team: sre - type: fluentd - service: fluentd-dsgpu-a-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -# - secretRef: -# name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: - - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - 04_outputs.conf: |- - diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/ingress-nginx-internal/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/ingress-nginx-internal/custom-values.yaml deleted file mode 100644 index 234c871..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/ingress-nginx-internal/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 6 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dsgpu-int-a-prd"}}}' - nodeSelector: - cloud.google.com/compute-class: c3-highcpu-4-nginx-internal-compute-class - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "c3-highcpu-4-nginx-internal-compute-class" - effect: "NoSchedule" diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/keda/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/keda/custom-values.yaml deleted file mode 100644 index 499f82f..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/keda/custom-values.yaml +++ /dev/null @@ -1,46 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - cloud.google.com/compute-class: datascience-devops - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - podLabels: - bu: "dsgpu" - team: "dsgpu-devops" - metricsAdapter: - bu: "dsgpu" - team: "dsgpu-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 300m - memory: 450Mi - # -- Manage [resource request & limits] of KEDA metrics apiserver pod - metricServer: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 100m - memory: 200Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 100m - memory: 180Mi - requests: - cpu: 60m - memory: 120Mi - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' \ No newline at end of file diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/kube-dns/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/kube-dns/custom-values.yaml deleted file mode 100644 index 79e3be1..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.1.64.2"],"prd.mrouter.int.svc.cluster.local":["10.1.64.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.1.64.2"]} diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/kube-events/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/kube-events/custom-values.yaml deleted file mode 100644 index 47ff986..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/kube-events/custom-values.yaml +++ /dev/null @@ -1,148 +0,0 @@ -fullnameOverride: kube-events-dsgpu-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - cloud.google.com/compute-class: datascience-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: datascience-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: datascience-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: datascience-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: datascience-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: datascience-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/kube-state-metrics/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 764ea6c..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-dsgpu-a-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-dsgpu-a-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "kube-state-metrics-dsgpu-a-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - cloud.google.com/compute-class: "c3-highcpu-44-vmselect-compute-class" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "c3-highcpu-44-vmselect-compute-class" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 17m - memory: 1.5Gi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap:z -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/kubectl-mcp-server/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/kubectl-mcp-server/custom-values.yaml deleted file mode 100644 index 7aefc0b..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/kubectl-mcp-server/custom-values.yaml +++ /dev/null @@ -1,121 +0,0 @@ -fullnameOverride: "kubectl-mcp-server" - -replicas: 1 - -image: - # In-house hardened build (helm-templates/kubectl-mcp-server/docker/) — - # NOT the upstream Docker Hub image. 266→18 HIGH/CRIT vulns, non-root, - # multi-stage slim base, kubectl v1.33.12 / helm v3.21.0. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/prd/devop/kubectl-mcp-server - tag: "v2" - pullPolicy: IfNotPresent - -labels: - bu: infra - team: devops - service: kubectl-mcp-server - env: prd - -serviceAccount: - create: true - annotations: {} - -rbac: - create: true - -podAnnotations: {} - -podSecurityContext: {} - -securityContext: - runAsNonRoot: true - runAsUser: 1000 - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -priorityClassName: "" - -# dsgpu is a GPU cluster with no `dedicated` nodepools — it uses GKE custom -# compute classes (NAP). The devops workload class is -# `datascience-devops`; nodes carry both the -# label and a matching NoSchedule taint (verified on k8s-dsgpu-prd-ase1). -nodeSelector: - cloud.google.com/compute-class: "datascience-devops" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: NoSchedule - -affinity: {} - -resources: - requests: - cpu: 250m - memory: 512Mi - limits: - cpu: 1000m - memory: 1Gi - -mcp: - mode: single - transport: http - host: "0.0.0.0" - port: 8000 - -auth: - allowAnonymous: false - -externalSecrets: - enabled: true - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/prd/cntr/devop/kubectl-mcp-server-dsgpu" - -# tcpSocket on purpose — streamable-http transport exposes only /mcp, -# there is no /health route (see chart values.yaml note). -livenessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - -readinessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - -service: - port: 8000 - type: ClusterIP - -# Contour standard (2.0.0-style): renders HTTPProxy parent/child + -intra -# instead of a networking.k8s.io Ingress. -createContourGateway: true - -ingress: - enabled: true - ingressClassName: contour-internal-1 - servicePortNumber: 8000 - hosts: - - host: kubectl-mcp-server-dsgpu.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/kyverno/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/kyverno/custom-values.yaml deleted file mode 100644 index aa931aa..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/kyverno/custom-values.yaml +++ /dev/null @@ -1,2244 +0,0 @@ -global: - - # -- Internal settings used with `helm template` to generate install manifest - # @ignored - templating: - enabled: false - debug: false - version: ~ - - image: - # -- (string) Global value that allows to set a single image registry across all deployments. - # When set, it will override any values set under `.image.registry` across the chart. - registry: ~ - # -- (list) Global list of Image pull secrets - # When set, it will override any values set under `imagePullSecrets` under different components across the chart. - imagePullSecrets: [] - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - caCertificates: - # -- Global CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - # Individual controller values will override this global value - data: ~ - - # -- Global value to set single volume to be mounted for CA certificates for all deployments. - # Not used when `.Values.global.caCertificates.data` is defined - # Individual controller values will override this global value - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Additional container environment variables to apply to all containers and init containers - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Global node labels for pod assignment. Non-global values will override the global value. - nodeSelector: - cloud.google.com/compute-class: dsgpu-kyverno - - # -- Global List of node taints to tolerate. Non-global values will override the global value. - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "dsgpu-kyverno" - effect: "NoSchedule" - -# -- (string) Override the name of the chart -nameOverride: ~ - -# -- (string) Override the expanded name of the chart -fullnameOverride: ~ - -# -- (string) Override the namespace the chart deploys to -namespaceOverride: ~ - -upgrade: - # -- Upgrading from v2 to v3 is not allowed by default, set this to true once changes have been reviewed. - fromV2: false - -apiVersionOverride: - # -- (string) Override api version used to create `PodDisruptionBudget`` resources. - # When not specified the chart will check if `policy/v1/PodDisruptionBudget` is available to - # determine the api version automatically. - podDisruptionBudget: ~ - -rbac: - roles: - # -- Aggregate ClusterRoles to Kubernetes default user-facing roles. For more information, see [User-facing roles](https://kubernetes.io/docs/reference/access-authn-authz/rbac/#user-facing-roles) - aggregate: - admin: true - view: true - -# Use openreports.io as the API group for reporting -openreports: - # -- Enable OpenReports feature in controllers - enabled: false - # -- Whether to install CRDs from the upstream OpenReports chart. Setting this to true requires enabled to also be true. - installCrds: false - -# CRDs configuration -crds: - - # -- Whether to have Helm install the Kyverno CRDs, if the CRDs are not installed by Helm, they must be added before policies can be created - install: true - - reportsServer: - # -- Kyverno reports-server is used in your cluster - enabled: false - - groups: - - # -- Install CRDs in group `kyverno.io` - kyverno: - cleanuppolicies: true - clustercleanuppolicies: true - clusterpolicies: true - globalcontextentries: true - policies: true - policyexceptions: true - updaterequests: true - - # -- Install CRDs in group `policies.kyverno.io` - policies: - validatingpolicies: true - policyexceptions: true - imagevalidatingpolicies: true - namespacedimagevalidatingpolicies: true - mutatingpolicies: true - generatingpolicies: true - deletingpolicies: true - namespaceddeletingpolicies: true - namespacedvalidatingpolicies: true - - # -- Install CRDs in group `reports.kyverno.io` - reports: - clusterephemeralreports: true - ephemeralreports: true - - # -- Install CRDs in group `wgpolicyk8s.io` - wgpolicyk8s: - clusterpolicyreports: false ##lk - policyreports: false ##lk - - # -- Additional CRDs annotations (Replace=true avoids 262144-byte annotation limit on large CRDs) - annotations: - argocd.argoproj.io/sync-options: Replace=true - # strategy.spinnaker.io/replace: 'true' - - # -- Additional CRDs labels - customLabels: {} - - migration: - - # -- Enable CRDs migration using helm post upgrade hook - enabled: true - - # -- Resources to migrate - resources: - - cleanuppolicies.kyverno.io - - clustercleanuppolicies.kyverno.io - - clusterpolicies.kyverno.io - - globalcontextentries.kyverno.io - - policies.kyverno.io - - policyexceptions.kyverno.io - - updaterequests.kyverno.io - - deletingpolicies.policies.kyverno.io - - generatingpolicies.policies.kyverno.io - - imagevalidatingpolicies.policies.kyverno.io - - namespacedimagevalidatingpolicies.policies.kyverno.io - - mutatingpolicies.policies.kyverno.io - - namespaceddeletingpolicies.policies.kyverno.io - - namespacedvalidatingpolicies.policies.kyverno.io - - policyexceptions.policies.kyverno.io - - validatingpolicies.policies.kyverno.io - - image: - # -- (string) Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- (string) Image repository - repository: kyverno/kyverno-cli - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- (string) Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podResources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -# Configuration -config: - - # -- Create the configmap. - create: true - - # -- Preserve the configmap settings during upgrade. - preserve: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - # -- Enable registry mutation for container images. Enabled by default. - enableDefaultRegistryMutation: true - - # -- The registry hostname used for the image mutation. - defaultRegistry: docker.io - - # -- Exclude groups - excludeGroups: - - system:nodes - - # -- Exclude usernames - excludeUsernames: [] - # - '!system:kube-scheduler' - - # -- Exclude roles - excludeRoles: [] - - # -- Exclude roles - excludeClusterRoles: [] - - # -- Generate success events. - generateSuccessEvents: false - - # -- Resource types to be skipped by the Kyverno policy engine. - # Make sure to surround each entry in quotes so that it doesn't get parsed as a nested YAML list. - # These are joined together without spaces, run through `tpl`, and the result is set in the config map. - # @default -- See [values.yaml](values.yaml) - resourceFilters: - - '[Event,*,*]' - # - '[*/*,kube-system,*]' - - '[*/*,kube-public,*]' - - '[*/*,kube-node-lease,*]' - - '[Node,*,*]' - - '[Node/?*,*,*]' - - '[APIService,*,*]' - - '[APIService/?*,*,*]' - - '[TokenReview,*,*]' - - '[SubjectAccessReview,*,*]' - - '[SelfSubjectAccessReview,*,*]' - - '[Binding,*,*]' - - '[Pod/binding,*,*]' - - '[ReplicaSet,*,*]' - - '[ReplicaSet/?*,*,*]' - - '[EphemeralReport,*,*]' - - '[ClusterEphemeralReport,*,*]' - # exclude resources from the chart - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:additional]' - - '[ClusterRoleBinding,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.configMapName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.metricsConfigMapName" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Job,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[Job/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.admission-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.background-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - # -- Sets the threshold for the total number of UpdateRequests generated for mutateExisitng and generate policies. - updateRequestThreshold: 1000 - - # -- Defines the `namespaceSelector`/`objectSelector` in the webhook configurations. - # The Kyverno namespace is excluded if `excludeKyvernoNamespace` is `true` (default) - webhooks: - # Exclude namespaces - # namespaceSelector: lk-s2 - # matchExpressions: - # - key: kubernetes.io/metadata.name - # operator: NotIn - # values: - # - kube-systemm lk-s2 - # Exclude objects - # objectSelector: - # matchExpressions: - # - key: webhooks.kyverno.io/exclude - # operator: DoesNotExist - - # -- Defines annotations to set on webhook configurations. - webhookAnnotations: - # Example to disable admission enforcer on AKS: - # 'admissions.enforcer/disabled': 'true' lk-s3 - - # -- Defines labels to set on webhook configurations. - webhookLabels: {} - # Example to adopt webhook resources in ArgoCD: - # 'argocd.argoproj.io/instance': 'kyverno' - - # -- Defines match conditions to set on webhook configurations (requires Kubernetes 1.27+). - matchConditions: [] - - # -- Exclude Kyverno namespace - # Determines if default Kyverno namespace exclusion is enabled for webhooks and resourceFilters - excludeKyvernoNamespace: true - - # -- resourceFilter namespace exclude - # Namespaces to exclude from the default resourceFilters - resourceFiltersExcludeNamespaces: [] - - # -- resourceFilters exclude list - # Items to exclude from config.resourceFilters - resourceFiltersExclude: [] - - # -- resourceFilter namespace include - # Namespaces to include to the default resourceFilters - resourceFiltersIncludeNamespaces: [] - - # -- resourceFilters include list - # Items to include to config.resourceFilters - resourceFiltersInclude: [] - -# Metrics configuration -metricsConfig: - - # -- Create the configmap. - create: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - namespaces: - - # -- List of namespaces to capture metrics for. - include: [] - - # -- list of namespaces to NOT capture metrics for. - exclude: [] - - # -- (string) Rate at which metrics should reset so as to clean up the memory footprint of kyverno metrics, if you might be expecting high memory footprint of Kyverno's metrics. Default: 0, no refresh of metrics. WARNING: This flag is not working since Kyverno 1.8.0 - metricsRefreshInterval: ~ - # metricsRefreshInterval: 24h - - # -- (list) Configures the bucket boundaries for all Histogram metrics, changing this configuration requires restart of the kyverno admission controller - bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 15, 20, 25, 30] - - # -- (map) Configures the exposure of individual metrics, by default all metrics and all labels are exported, changing this configuration requires restart of the kyverno admission controller - metricsExposure: - kyverno_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_image_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_mutating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_generating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_admission_review_duration_seconds: - # enabled: false - disabledLabelDimensions: ["resource_namespace"] - kyverno_policy_rule_info_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_policy_results_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_admission_requests_total: - disabledLabelDimensions: ["resource_namespace"] - kyverno_cleanup_controller_deletedobjects_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - -# -- Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -imagePullSecrets: {} - # regcred: - # registry: foo.example.com - # username: foobar - # password: secret - # regcred2: - # registry: bar.example.com - # username: barbaz - # password: secret2 - -# -- Existing Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -existingImagePullSecrets: [] - # - test-registry - # - other-test-registry - -# Tests configuration -test: - # -- Sleep time before running test - sleep: 20 - - image: - # -- (string) Image registry - registry: curlimages - # -- Image repository - repository: curl - # -- Image tag - # Defaults to `latest` if omitted - tag: '8.10.1' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - # -- Security context for the test containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- Additional Pod annotations - podAnnotations: {} - - # -- List of node taints to tolerate - tolerations: [] - -# -- Additional labels -customLabels: {} - - -webhooksCleanup: - # -- Create a helm pre-delete hook to cleanup webhooks. - enabled: true - - autoDeleteWebhooks: - # -- Allow webhooks controller to delete webhooks using finalizers - enabled: false - - image: - # -- (string) Image registry - registry: registry.k8s.io - # -- Image repository - repository: kubectl - # -- Image tag - # Defaults to `latest` if omitted - tag: 'v1.32.7' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -grafana: - # -- Enable grafana dashboard creation. - enabled: false - - # -- Configmap name template. - configMapName: '{{ include "kyverno.fullname" . }}-grafana' - - # -- (string) Namespace to create the grafana dashboard configmap. - # If not set, it will be created in the same namespace where the chart is deployed. - namespace: ~ - - # -- Grafana dashboard configmap annotations. - annotations: {} - - # -- Grafana dashboard configmap labels - labels: - grafana_dashboard: "1" - - # -- create GrafanaDashboard custom resource referencing to the configMap. - # according to https://grafana-operator.github.io/grafana-operator/docs/examples/dashboard_from_configmap/readme/ - grafanaDashboard: - create: false - folder: kyverno - allowCrossNamespaceImport: true - matchLabels: - dashboards: "grafana" - -# Features configuration -features: - admissionReports: - # -- Enables the feature - enabled: true - aggregateReports: - # -- Enables the feature - enabled: true - policyReports: - # -- Enables the feature - enabled: false ##lk - validatingAdmissionPolicyReports: - # -- Enables the feature - enabled: true - mutatingAdmissionPolicyReports: - # -- Enables the feature - enabled: false - reporting: - # -- Enables the feature - validate: true - # -- Enables the feature - mutate: true - # -- Enables the feature - mutateExisting: true - # -- Enables the feature - imageVerify: true - # -- Enables the feature - generate: true - autoUpdateWebhooks: - # -- Enables the feature - enabled: true - backgroundScan: - # -- Enables the feature - enabled: true - # -- Number of background scan workers - backgroundScanWorkers: 2 - # -- Background scan interval - backgroundScanInterval: 1h - # -- Skips resource filters in background scan - skipResourceFilters: true - configMapCaching: - # -- Enables the feature - enabled: true - controllerRuntimeMetrics: - # -- Bind address for controller-runtime metrics (use "0" to disable it) - bindAddress: ":8080" - deferredLoading: - # -- Enables the feature - enabled: true - dumpPayload: - # -- Enables the feature - enabled: false - forceFailurePolicyIgnore: - # -- Enables the feature - enabled: false - generateValidatingAdmissionPolicy: - # -- Enables the feature - enabled: true - generateMutatingAdmissionPolicy: - # -- Enables the feature - enabled: false - dumpPatches: - # -- Enables the feature - enabled: false - globalContext: - # -- Maximum allowed response size from API Calls. A value of 0 bypasses checks (not recommended) - maxApiCallResponseLength: 2000000 - logging: - # -- Logging format - format: text - # -- Logging verbosity - verbosity: 2 - omitEvents: - # -- Events which should not be emitted (possible values `PolicyViolation`, `PolicyApplied`, `PolicyError`, and `PolicySkipped`) - eventTypes: - - PolicyApplied - - PolicySkipped - # - PolicyViolation - # - PolicyError - policyExceptions: - # -- Enables the feature - enabled: true - # -- Restrict policy exceptions to a single namespace - # Set to "*" to allow exceptions in all namespaces - namespace: '' - protectManagedResources: - # -- Enables the feature - enabled: false - registryClient: - # -- Allow insecure registry - allowInsecure: false - # -- Enable registry client helpers - credentialHelpers: - - default - - google - - amazon - - azure - - github - ttlController: - # -- Reconciliation interval for the label based cleanup manager - reconciliationInterval: 1m - tuf: - # -- Enables the feature - enabled: false - # -- (string) Path to Tuf root - root: ~ - # -- (string) Raw Tuf root - rootRaw: ~ - # -- (string) Tuf mirror - mirror: ~ - -# Admission controller configuration -admissionController: - autoscaling: - # -- Enable horizontal pod autoscaling - enabled: true - - # -- Minimum number of pods - minReplicas: 1 - - # -- Maximum number of pods - maxReplicas: 10 - - # -- Target CPU utilization percentage - targetCPUUtilizationPercentage: 80 - - # -- Configurable scaling behavior - behavior: {} - - # -- Overrides features defined at the root level - featuresOverride: - admissionReports: - # -- Max number of admission reports allowed in flight until the admission controller stops creating new ones - backPressureThreshold: 1000 - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- The ServiceAccount name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno admission controller activities. - # This will help ensure Kyverno stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- admissionController webhook server port - # in case you are using hostNetwork: true, you might want to change the port the webhookServer is listening to - webhookServer: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - admission-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.admissionController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - initContainer: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyvernopre - # -- (string) Image tag - # If missing, defaults to image.tag - tag: ~ - # -- (string) Image pull policy - # If missing, defaults to image.pullPolicy - pullPolicy: ~ - - resources: - # -- Pod resource limits - limits: - cpu: 1000m - memory: 1Gi - # -- Pod resource requests - requests: - cpu: 1000m - memory: 1Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - container: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyverno - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - resources: - # -- Pod resource limits - limits: - cpu: 1900m - memory: 3Gi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Array of extra init containers - extraInitContainers: [] - # - name: init-container - # image: busybox - # command: ['sh', '-c', 'echo Hello'] - - # -- Array of extra containers to run alongside kyverno - extraContainers: [] - # - name: myapp-container - # image: busybox - # command: ['sh', '-c', 'echo Hello && sleep 3600'] - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Kyverno's metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Background controller configuration -backgroundController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable background controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: - - apiGroups: - - networking.k8s.io - resources: - - ingresses - - ingressclasses - - networkpolicies - verbs: - - create - - update - - patch - - delete - - apiGroups: - - rbac.authorization.k8s.io - resources: - - rolebindings - - roles - verbs: - - create - - update - - patch - - delete - - apiGroups: - - '' - resources: - - configmaps - - resourcequotas - - limitranges - verbs: - - create - - update - - patch - - delete - - apiGroups: - - resource.k8s.io - resources: - - resourceclaims - - resourceclaimtemplates - verbs: - - create - - delete - - update - - patch - - deletecollection - - apiGroups: - - autoscaling - resources: - - horizontalpodautoscalers - verbs: - - get - - list - - watch - - update - - patch - - apiGroups: - - apps - resources: - - deployments - - statefulsets - - replicasets - verbs: - - get - - list - - watch - - update - - patch - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - # - patch - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/background-controller - # -- Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 1900m - memory: 3Gi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - background-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.backgroundController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - # -- backgroundController server port - # in case you are using hostNetwork: true, you might want to change the port the backgroundController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Cleanup controller configuration -cleanupController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable cleanup controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - delete - # - list - # - watch - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/cleanup-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- cleanupController server port - # in case you are using hostNetwork: true, you might want to change the port the cleanupController is listening to - server: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 1100m - memory: 1.2Gi - # -- Pod resource requests - requests: - cpu: 1100m - memory: 1.2Gi - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - cleanup-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `service.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Reports controller configuration -reportsController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable reports controller. - enabled: false ##lk - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/reports-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno reports controller activities. - # This will help ensure Kyverno reports stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 200m - memory: 128Mi - # -- Pod resource requests - requests: - cpu: 100m - memory: 64Mi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - reports-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.reportsController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- (string) Service node port. - # Only used if `type` is `NodePort`. - nodePort: ~ - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- (string) Traces receiver address - address: ~ - # -- (string) Traces receiver port - port: ~ - # -- (string) Traces receiver credentials - creds: ~ - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- (string) Otel collector endpoint - collector: ~ - # -- (string) Otel collector credentials - creds: ~ - - # -- reportsController server port - # in case you are using hostNetwork: true, you might want to change the port the reportsController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - - # -- Enable sanity check for reports CRDs - sanityChecks: true diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/kyverno/policies/protect-namespace.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/kyverno/policies/protect-namespace.yaml deleted file mode 100644 index 460e19d..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/kyverno/policies/protect-namespace.yaml +++ /dev/null @@ -1,40 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: protect-critical-namespaces -spec: - background: false - rules: - - name: deny-deletion-of-critical-namespaces - match: - any: - - resources: - kinds: - - Namespace - names: - - keda-demand-prd - - contour-internal-0-demand-prd - - contour-internal-0-demand-prd-intra - - contour-external-demand-prd - - external-secrets-demand-prd - - flagger-demand-prd - - victoriametrics - - kube-system - - monitoring - - telegraf-operator - - loadtester - - opentelemetry - - fluentd - - kube-events - - observability - - contour-cert-checker-ns - - aurva-dataplane - validate: - failureAction: Enforce - message: "Deletion of critical namespaces is not allowed." - deny: - conditions: - any: - - key: "{{request.operation}}" - operator: Equals - value: DELETE diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/kyverno/policies/restrict-replicas.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/kyverno/policies/restrict-replicas.yaml deleted file mode 100644 index 101bbeb..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/kyverno/policies/restrict-replicas.yaml +++ /dev/null @@ -1,29 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: restrict-minimum-replicas-selected-ns -spec: - background: false - rules: - - name: min-replicas-deployments-statefulsets - match: - resources: - kinds: - - Deployment - - StatefulSet - - Deployment/scale - - StatefulSet/scale - names: - - coredns - namespaces: - - kube-system - operations: - - CREATE - - UPDATE - validate: - failureAction: Enforce - message: "Deployments and StatefulSets must have at least 5 replicas in this namespace." - anyPattern: - # Case 1: replicas explicitly set and >= 5 - - spec: - replicas: ">=5" diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/loadtester/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/loadtester/custom-values.yaml deleted file mode 100644 index a99537a..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/loadtester/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -replicaCount: 1 - -Namespace: loadtester - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/dev/devops/flagger-loadtester - tag: 0.30.0 - pullPolicy: IfNotPresent - pullSecret: - -podLabels: - bu: dsgpu - team: dsgpu-devops - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - -podPriorityClassName: "" - -logLevel: info -cmd: - timeout: 1h - namespaceRegexp: "" - -nameOverride: "loadtester" -fullnameOverride: "" - -env: [] - -service: - type: ClusterIP - port: 80 - -resources: - requests: - cpu: 10m - memory: 64Mi - -volumes: [] -volumeMounts: [] - -nodeSelector: - cloud.google.com/compute-class: datascience-devops - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - -affinity: {} - -ingress: - slowStart: - enabled: false - enabled: true - ingressClassName: contour-internal-1 - servicePort: http - hosts: - - host: dsgpu-flagger-loadtester.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / -createContourGateway: true -contourResponseTimeout: false -rbac: - # rbac.create: `true` if rbac resources should be created - create: true - # rbac.scope: `cluster` to create cluster-scope rbac resources (ClusterRole/ClusterRoleBinding) - # otherwise, namespace-scope rbac resources will be created (Role/RoleBinding) - scope: - # rbac.rules: array of rules to apply to the role. example: - # rules: - # - apiGroups: [""] - # resources: ["pods"] - # verbs: ["list", "get"] - rules: [] - -# name of an existing service account to use - if not creating rbac resources -serviceAccountName: "" - -# App Mesh virtual node settings (to be used for AppMesh v1beta1) -meshName: "" -#backends: -# - app1.namespace -# - app2.namespace - -# App Mesh virtual node settings (to be used for AppMesh v1beta2) -appmesh: - enabled: false - backends: - - podinfo - - podinfo-canary - -#Istio virtual service and gatway settings. TLS secrets should be in namespace before enbaled it. ( secret format loadtester.fullname ) -istio: - enabled: false - host: flagger-loadtester.flagger - gateway: - enabled: false - tls: - enabled: false - httpsRedirect: false - -# when enabled, it will add a security context for the loadtester pod -securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 100 - runAsGroup: 101 - -podDisruptionBudget: - enabled: false - minAvailable: 1 diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/opentelemetry-coralogix/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/opentelemetry-coralogix/custom-values.yaml deleted file mode 100644 index d96164d..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/opentelemetry-coralogix/custom-values.yaml +++ /dev/null @@ -1,281 +0,0 @@ -global: - domain: "coralogixsg.com" - defaultApplicationName: "default" - defaultSubsystemName: "nodes" - - # Old endpoint based configuration, - # please use domain instead. - traces: - endpoint: "" - metrics: - endpoint: "" - logs: - endpoint: "" - -# set distribution to openshift for openshift clusters -distribution: "" - -# Opentelemetry collector configuration -dedicatedValue: false -schedulerName: default-scheduler - -labels: - bu: "dsgpu" - team: "sre" - service: "opentelemetry-dsgpu-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: false - key: prd/admin/coralogix-keys - secretStoreRef: - name: vault-backend - -image: - # If you want to use the core image `otel/opentelemetry-collector`, you also need to change `command.name` value to `otelcol`. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - # repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - tag: "0.87.0" - # When digest is set to a non-empty value, images will be pulled by digest (regardless of tag value). - digest: "" -mode: daemonset -rollout: - rollingUpdate: - # When 'mode: daemonset', maxSurge cannot be used when hostPort is set for any of the ports - # maxSurge: 25% - maxUnavailable: 1 - strategy: RollingUpdate -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" -hostNetwork: true -dnsPolicy: "ClusterFirstWithHostNet" -fullnameOverride: opentelemetry-dsgpu-prd - -presets: - logsCollection: - enabled: false - storeCheckpoints: false - maxRecombineLogSize: 1048576 - extraFilelogOperators: [] -# - type: recombine -# combine_field: body -# source_identifier: attributes["log.file.path"] -# is_first_entry: body matches "^(YOUR-LOGS-REGEX)" - kubernetesAttributes: - enabled: false - hostMetrics: - enabled: false - kubeletMetrics: - enabled: false - -extraEnvs: -- name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" -- name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName -config: - extensions: {} - # zpages: - # endpoint: localhost:55679 - # pprof: - # endpoint: localhost:1777 - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 4s - tls: - insecure: true - sending_queue: - enabled: true - queue_size: 6000 - num_consumers: 200 - resolver: - dns: - hostname: opentelemetry-admin-prd.opentelemetry.svc.clusterset.local - # coralogix: - # timeout: "30s" - # private_key: "${CORALOGIX_PRIVATE_KEY}" - # domain: "{{.Values.global.domain}}" - # traces: - # endpoint: "{{ .Values.global.traces.endpoint }}" - # metrics: - # endpoint: "{{ .Values.global.metrics.endpoint }}" - # logs: - # endpoint: "{{ .Values.global.logs.endpoint }}" - # application_name_attributes: - # - "k8s.namespace.name" - # - "service.namespace" - # subsystem_name_attributes: - # - "k8s.deployment.name" - # - "k8s.statefulset.name" - # - "k8s.daemonset.name" - # - "k8s.cronjob.name" - # - "k8s.job.name" - # - "k8s.container.name" - # - "k8s.node.name" - # - "service.name" - # application_name: "{{.Values.global.defaultApplicationName }}" - # subsystem_name: "{{.Values.global.defaultSubsystemName }}" - processors: - k8sattributes: - filter: - node_from_env_var: KUBE_NODE_NAME - extract: - metadata: - - "k8s.namespace.name" - - "k8s.deployment.name" - - "k8s.statefulset.name" - - "k8s.daemonset.name" - - "k8s.cronjob.name" - - "k8s.job.name" - - "k8s.pod.name" - - "k8s.node.name" - memory_limiter: null # Will get the k8s resource limits - resourcedetection/env: - detectors: ["system","env"] - timeout: 2s - override: false - # spanmetrics: - # metrics_exporter: coralogix - # dimensions: - # - name: "k8s.deployment.name" - # - name: "k8s.statefulset.name" - # - name: "k8s.daemonset.name" - # - name: "k8s.cronjob.name" - # - name: "k8s.job.name" - # - name: "k8s.container.name" - # - name: "k8s.node.name" - # - name: "k8s.namespace.name" - receivers: - otlp: - protocols: - grpc: - endpoint: ${MY_POD_IP}:4317 - http: - endpoint: ${MY_POD_IP}:4318 - zipkin: - endpoint: ${MY_POD_IP}:9411 - jaeger: - protocols: - grpc: - endpoint: ${MY_POD_IP}:14250 - thrift_http: - endpoint: ${MY_POD_IP}:14268 - thrift_compact: - endpoint: ${MY_POD_IP}:6831 - thrift_binary: - endpoint: ${MY_POD_IP}:6832 - prometheus: - config: - scrape_configs: - - job_name: opentelemetry-collector - scrape_interval: 30s - static_configs: - - targets: - - ${MY_POD_IP}:8888 - service: - extensions: - # - zpages - # - pprof - - health_check -# - memory_ballast - telemetry: - metrics: - address: ${env:MY_POD_IP}:8888 - pipelines: - traces: - exporters: - - loadbalancing - processors: - - memory_limiter - # - spanmetrics - - batch - receivers: - - otlp - metrics: null - logs: null -tolerations: - - operator: Exists - -resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 1 - memory: 2G - -ports: - jaeger-binary: - enabled: true - containerPort: 6832 - servicePort: 6832 - hostPort: 6832 - protocol: TCP - # In order to enable podMonitor, following part must be enabled in order to expose the required port: - # metrics: - # enabled: true - -# podMonitor: -# enabled: true - -# prometheusRule: -# enabled: true -# defaultRules: -# enabled: true - -# Annotations to be added to pod -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-dsgpu-contour-ext - operator: NotIn - values: - - dedicated - - key: p-dsgpu-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-dsgpu-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-dsgpu-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-dsgpu-default-prd-ase1 - - key: cloud.google.com/compute-class - operator: NotIn - values: - - n4d-highcpu-64-vmagent-compute-class - - n4-highcpu-16-vminsert-compute-class - - c3-highcpu-44-vmselect-compute-class - - n4d-highmem-48-vmstorage-compute-class \ No newline at end of file diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index 86578f3..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,134 +0,0 @@ -fullnameOverride: opentelemetry-dsgpu-prd - -mode: daemonset - -# priorityClassName: "system-node-critical" -# Removing since dataengg pods are going into pending state -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-dsgpu-contour-ext - operator: NotIn - values: - - dedicated - - key: p-dsgpu-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-dsgpu-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-dsgpu-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-dsgpu-default-prd-ase1 - - key: cloud.google.com/compute-class - operator: NotIn - values: - - n4d-highcpu-64-vmagent-compute-class - - n4-highcpu-16-vminsert-compute-class - - c3-highcpu-44-vmselect-compute-class - - n4d-highmem-48-vmstorage-compute-class - - c3-highcpu-22-contour-internal-1-compute-class - - c4d-highcpu-16-contour-internal-1-compute-class - - c3-highcpu-22-contour-internal-0-compute-class - - c4d-highcpu-16-contour-internal-0-compute-class - -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 100m - memory: 124Mi - limits: - cpu: 200m - memory: 256Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "dsgpu" - team: "sre" - service: "opentelemetry-dsgpu-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/paused-container/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/paused-container/custom-values.yaml deleted file mode 100644 index df5354c..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/paused-container/custom-values.yaml +++ /dev/null @@ -1,82 +0,0 @@ -# Default values for hello-world. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -priorityClass: - name: "low-priority" - -image: - repository: nginx - tag: "1.14.2" - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - -nameOverride: "" -fullnameOverride: "paused-container-dsgpu-prd" - -extraLabels: - team: "devops" - bu: "dsgpu" - env: "prd" - service: "paused-container-dsgpu-prd" - priority: "p3" - type: "tools" - -topologySpreadConstraints: [] - # - labelSelector: - # matchLabels: - # dedicated: vminsert - # maxSkew: 1 - # topologyKey: topology.kubernetes.io/hostname - # whenUnsatisfiable: DoNotSchedule - -affinity: - podAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - - labelSelector: - matchExpressions: - - key: type - operator: In - values: - - vmselect - - vminsert - - tools - topologyKey: topology.kubernetes.io/zone - podAntiAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - - labelSelector: - matchExpressions: - - key: type - operator: In - values: - - vmselect - - vminsert - - tools - topologyKey: kubernetes.io/hostname - namespaceSelector: {} - - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" - -service: - type: ClusterIP - port: 80 - -deployments: - - nodepool: vminsert - bu: dsgpu - cpuRequest: 1 - memoryRequest: 1Gi - replicaCount: 3 - - nodepool: vmselect - bu: dsgpu - cpuRequest: 1 - memoryRequest: 1Gi - replicaCount: 3 diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/prometheus-node-exporter/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index dac2b84..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,496 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus' global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric's labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job's name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-dataengg-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "dsgpu" - team: "sre" - service: "node-exporter-dsgpu-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index 862e026..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-dsgpu-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-dsgpu-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dsgpu" - team: "sre" - service: "stackdriver-exporter-dsgpu-prd" - env: "prd" - priority: "p0" - type: "exporter" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-dsgpu-prd-0622" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,logging.googleapis.com/user,run.googleapis.com/,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime,compute.googleapis.com/instance' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - cloud.google.com/compute-class: c3-highcpu-44-vmselect-compute-class - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "c3-highcpu-44-vmselect-compute-class" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-dssre-stackdriver-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/telegraf-operator/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/telegraf-operator/custom-values.yaml deleted file mode 100644 index c7298e2..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,150 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "telegraf-operator-dsgpu-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - cloud.google.com/compute-class: "c3-highcpu-44-vmselect-compute-class" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "c3-highcpu-44-vmselect-compute-class" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml deleted file mode 100644 index 9bceac4..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml +++ /dev/null @@ -1,272 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 1 - -fullnameOverride: vmagent-dsgpu-a-prd-fb -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-dsgpu-a-prd-fb-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - -# Remove topology spread constraints -topologySpreadConstraints: [] - -# Modify deployment strategy -deployment: - enabled: true - -# Remove HPA -horizontalPodAutoscaler: - enabled: false - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-infr-sre-vmagent-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -# Remove pod disruption budget -podDisruptionBudget: - enabled: false - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-dsgpu.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vmsinglenode-infra-prd.victoriametrics.svc.clusterset.local:8428/api/v1/write - - https://vmsinglenode-infra-prd.meeshogcp.in/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dsgpu-fb" - team: "sre" - service: "vmagent-dsgpu-a-prd-fb" - env: "prd" - priority: "p0" - type: "vmagent-fb" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dsgpu-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dsgpu-a-prd-fb.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -# Modify resources -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - requests: - cpu: 2 - memory: 1Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - cloud.google.com/compute-class: n4d-highcpu-64-vmagent-compute-class - -tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: n4d-highcpu-64-vmagent-compute-class - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml deleted file mode 100644 index ed7cf09..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-dsgpu-a-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-dsgpu-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-dsgpu-a-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-dsgpu-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/dsgpu/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-dsgpu-a-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "dsgpu" - team: "sre" - service: "vmalert-dsgpu-a-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "c3-highcpu-44-vmselect-compute-class" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "c3-highcpu-44-vmselect-compute-class" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/victoriametrics-agent/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index 96407e4..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,480 +0,0 @@ -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "dsgpu-a-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -# -- Replica count -replicaCount: 2 - -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - -# -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables -deployment: - spec: - minReadySeconds: 1200 - progressDeadlineSeconds: 1800 - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - -image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - # -- Image tag, set to `Chart.AppVersion` by default - tag: v1.133.0 # rewrites Chart.AppVersion - # -- Variant of the image to use. - # e.g. enterprise, scratch - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - -# -- Image pull secrets -imagePullSecrets: [] - -# -- Add additional DNS entries to pods hosts file. Check [official documentation](https://kubernetes.io/docs/tasks/network/customize-hosts-file-for-pods/) -hostAliases: [] -# - ip: 192.168.1.1 -# hostNames: -# - test.example.com -# - another.example.net - -# -- Override chart name -nameOverride: "" - -# -- Override resources fullname -fullnameOverride: "vm-agent-dsgpu-a-prd" - -# -- Container working directory -containerWorkingDir: "/" - -rbac: - # -- Enables Role/RoleBinding creation - create: true - - # -- Role/RoleBinding annotations - annotations: {} - - # -- Role/RoleBinding labels - extraLabels: {} - - # -- If true and `rbac.enabled`, will deploy a Role/RoleBinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - - # -- additional rules for a role - extraRules: [] - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - # -- Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-sre-vmagnt-prd-mds@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # -- mount API token to pod directly - automountToken: true - -# -- See `kubectl explain poddisruptionbudget.spec` for more or check [official documentation](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# -- Generates `remoteWrite.*` flags and config maps with value content for values, that are of type list of map. -# Each item should contain `url` param to pass validation. -remoteWrite: - - url: http://vm-insert-dsgpu-a-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write - disableOnDiskQueue: false - dropSamplesOnOverload: false - # - url: http://vm-insert-dsgpu-prd.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/api/v1/write - # disableOnDiskQueue: false - # dropSamplesOnOverload: false - -# -- VMAgent extra command line arguments -extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8429 - promscrape.config.strictParse: false # default is true - promscrape.maxScrapeSize: 1000000000 # default is 167772160 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - # promscrape.suppressScrapeErrors: true # default is false - - # Uncomment and specify the port if you want to support any of the protocols: - # https://docs.victoriametrics.com/victoriametrics/vmagent/#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for more details. -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth-secret - # key: password - -# -- Specify alternative source for env variables -envFrom: - [] - #- configMapRef: - # name: special-config - -# -- Extra labels for Deployment and Statefulset -extraLabels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "vm-agent-dsgpu-a-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra labels for Pods only -podLabels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "vm-agent-dsgpu-a-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra selector labels common for pod and service -selectorLabels: {} - -# -- Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# -- Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# -- Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -# -- Extra containers to run in a pod with vmagent -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -# -- Init containers for vmagent -initContainers: - [] - # - name: example - # image: example-image - -# -- Security context to be added to pod -podSecurityContext: - enabled: true - # fsGroup: 2000 - -# -- Security context to be added to pod's containers -securityContext: - enabled: true - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - # -- Enable agent service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - extraLabels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) for details - externalIPs: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8429 - # -- Target port - targetPort: http - # nodePort: 30000 - # -- Service type - type: ClusterIP - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - -ingress: - # -- Enable deployment of ingress for agent - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-agent-dsgpu-a-prd.meeshogcp.in - path: - - / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - - # -- Ingress controller class name - ingressClassName: "nginx-internal" - - # -- Ingress path type - pathType: Prefix - -# -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) -resources: - requests: - cpu: 58 - memory: 110Gi - -# -- Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) -nodeSelector: - cloud.google.com/compute-class: "n4d-highcpu-64-vmagent-compute-class" - -# -- Node tolerations for server scheduling to nodes with taints. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "n4d-highcpu-64-vmagent-compute-class" - effect: "NoSchedule" - -# -- Pod topologySpreadConstraints -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# -- Pod affinity -affinity: {} - -# -- VMAgent [scraping configuration](https://docs.victoriametrics.com/victoriametrics/vmagent/#how-to-collect-metrics-in-prometheus-format) -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-dsgpu-a-prd" - -# -- Priority class to be assigned to the pod(s) -priorityClassName: "" - -# -- Enable the host network -hostNetwork: false - -serviceMonitor: - # -- Enable deployment of Service Monitor for server component. This is Prometheus operator object - enabled: false - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Service Monitor relabelings - relabelings: [] - # -- Basic auth params for Service Monitor - basicAuth: {} - # -- Service Monitor metricRelabelings - metricRelabelings: [] - # -- Service Monitor targetPort - targetPort: http - # interval: 15s - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - -# -- Empty dir configuration for a case, when persistence is disabled -emptyDir: {} - -persistentVolume: - # -- Create/use Persistent Volume Claim for server component. Empty dir if false - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Size of the volume. Should be calculated based on the logs you send and retention policy you set. - size: 10Gi - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume additional labels - extraLabels: {} - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Horizontal Pod Autoscaling. -# Note that it is not intended to be used for vmagents which perform scraping. -# In order to scale scraping vmagents check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) -horizontalPodAutoscaling: - # -- Use HPA for vmagent - enabled: false - # -- Maximum replicas for HPA to use to to scale vmagent - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale vmagent - minReplicas: 1 - # -- Metric for HPA to use to scale vmagent - metrics: [] - -# -- VMAgent scrape configuration -config: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -probe: - # -- Readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - # -- Liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - # -- Startup probe - startup: {} - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -allowedMetricsEndpoints: - - /metrics - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/victoriametrics-insert/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/victoriametrics-insert/custom-values.yaml deleted file mode 100644 index d31933e..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/victoriametrics-insert/custom-values.yaml +++ /dev/null @@ -1,411 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: false - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- IDs of vmstorage nodes to exclude from writing - excludeStorageIDs: [] - # -- Override default `app` label name - name: vminsert - # -- VMInsert strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - # override Chart.AppVersion - tag: v1.107.0-cluster - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMInsert http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-dsgpu-a-prd - # -- Extra command line arguments for vminsert component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8480 - maxLabelsPerTimeseries: 60 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-dsgpu-a-prd-0.vm-storage-dsgpu-a-prd.victoriametrics.svc:8400" - - "vm-storage-dsgpu-a-prd-1.vm-storage-dsgpu-a-prd.victoriametrics.svc:8400" - - "vm-storage-dsgpu-a-prd-2.vm-storage-dsgpu-a-prd.victoriametrics.svc:8400" - - "vm-storage-dsgpu-a-prd-3.vm-storage-dsgpu-a-prd.victoriametrics.svc:8400" - - "vm-storage-dsgpu-a-prd-4.vm-storage-dsgpu-a-prd.victoriametrics.svc:8400" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "vm-insert-dsgpu-a-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Pod's additional labels - podLabels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "vm-insert-dsgpu-a-prd" - env: "prd" - priority: "p0" - type: "vminsert" - terminationGracePeriodSeconds: 30 - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - - # -- Readiness & Liveness probes - probe: - # -- VMInsert readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert startup probe - startup: {} - - # -- Relabel configuration - relabel: - enabled: false - config: [] - # -- Use existing configmap if specified - # otherwise .config values will be used. Relabel config **should** reside under `relabel.yml` key - configMap: "" - - # Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 7 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vminsert - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vminsert - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "n4-highcpu-16-vminsert-compute-class" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#nodeselector) - nodeSelector: - cloud.google.com/compute-class: "n4-highcpu-16-vminsert-compute-class" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 10 - memory: 10Gi - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - service: - # -- Create VMInsert service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here]( https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) for details - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-insert-dsgpu-a-prd.meeshogcp.in - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - - # -- Ingress controller class name - ingressClassName: nginx-internal - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for insert component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/insert" .Values.vminsert }}' - - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vminsert component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/victoriametrics-select/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/victoriametrics-select/custom-values.yaml deleted file mode 100644 index f55a4d7..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/victoriametrics-select/custom-values.yaml +++ /dev/null @@ -1,444 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - splitService: true - externalService: - enabled: true - name: vmselect-dsgpu-a-prd-proxy - # -- Override default `app` label name - clusternativeService: - enabled: false - name: "" - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # override Chart.AppVersion - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMSelect http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-dsgpu-a-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - # -- Extra command line arguments for vmselect component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8481 - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-dsgpu-a-prd-0.vm-storage-dsgpu-a-prd.victoriametrics.svc:8401" - - "vm-storage-dsgpu-a-prd-1.vm-storage-dsgpu-a-prd.victoriametrics.svc:8401" - - "vm-storage-dsgpu-a-prd-2.vm-storage-dsgpu-a-prd.victoriametrics.svc:8401" - - "vm-storage-dsgpu-a-prd-3.vm-storage-dsgpu-a-prd.victoriametrics.svc:8401" - - "vm-storage-dsgpu-a-prd-4.vm-storage-dsgpu-a-prd.victoriametrics.svc:8401" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "vm-select-dsgpu-a-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # -- Pod's additional labels - podLabels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "vm-select-dsgpu-a-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Readiness & Liveness probes - probe: - # -- VMSelect readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 10 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vmselect - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vmselect - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "c3-highcpu-44-vmselect-compute-class" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - cloud.google.com/compute-class: "c3-highcpu-44-vmselect-compute-class" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 38 - memory: 70Gi - - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Create VMSelect service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: false - - # -- Ingress annotations - annotations: {} - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-select-dsgpu-a.prd.meesho.int - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # -- Ingress controller class name - ingressClassName: contour-internal-0 - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for select component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/select" .Values.vmselect }}' - - # -- vmselect mode: deployment, daemonSet - mode: deployment - - # -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables - deployment: - spec: - # -- VMSelect strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - - # -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables - statefulSet: - enabled: false - spec: - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume extra labels - extraLabels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmselect component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: false - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false diff --git a/helm-overrides/gke-dsgpu-prd-ase1a/victoriametrics-storage/custom-values.yaml b/helm-overrides/gke-dsgpu-prd-ase1a/victoriametrics-storage/custom-values.yaml deleted file mode 100644 index c092ae2..0000000 --- a/helm-overrides/gke-dsgpu-prd-ase1a/victoriametrics-storage/custom-values.yaml +++ /dev/null @@ -1,363 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - enabled: false - -vminsert: - enabled: false - -vmauth: - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- Override default `app` label name - name: vmstorage - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Variant of the image to use. e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMStorage http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-dsgpu-a-prd - - # -- Additional environment variables (ex.: secret tokens, flags). Check https://docs.victoriametrics.com/victoriametrics/#environment-variables for details - env: [] - # -- Specify alternative source for env variables - envFrom: [] - # -- Data retention period. Possible units character: h(ours), d(ays), w(eeks), y(ears), if no unit character specified - month. The minimum retention period is 24h. - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - httpListenAddr: :8482 - loggerTimezone: "Asia/Kolkata" - dedup.minScrapeInterval: 60s - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "n4d-highmem-48-vmstorage-compute-class" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - cloud.google.com/compute-class: "n4d-highmem-48-vmstorage-compute-class" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - # -- Use an alternate scheduler, e.g. "stork". Check https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ for details - # schedulerName: - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Override Persistent Volume Claim name - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are https://kubernetes.io/docs/concepts/storage/persistent-volumes/ - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume extra labels - extraLabels: {} - # -- Storage class name. Will be empty if not set - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume - size: 8134Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "vm-storage-dsgpu-a-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - # -- Pod's additional labels - podLabels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "vm-storage-dsgpu-a-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Details are https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 42 - memory: 330Gi - - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: - enabled: false - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - podSecurityContext: - enabled: false - service: - enabled: true - # -- Service annotations - annotations: {} - # -- Service ClusterIP - clusterIP: None - # -- Service type - type: ClusterIP - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmselect - vmselectPort: 8401 - # -- Extra service ports - extraPorts: [] - # -- Health check node port for a service. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilyPolicy: "" - # -- List of service IP families. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilies: [] - # -- Traffic Distribution. Check https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution - trafficDistribution: "" - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - minReadySeconds: 5 - # -- Readiness probes - probe: - # -- VMStorage readiness probe - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMStorage startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmstorage component - enabled: false - # -- Maximum replicas for HPA to use to to scale the vmstorage component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmstorage component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmstorage component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: - scaleDown: - selectPolicy: Disabled - - vmbackupmanager: - # -- Enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-farmiso-prd-ase1a/README.md b/helm-overrides/gke-farmiso-prd-ase1a/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/gke-farmiso-prd-ase1a/aurva-dataplane/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/aurva-dataplane/custom-values.yaml deleted file mode 100644 index b51e431..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/aurva-dataplane/custom-values.yaml +++ /dev/null @@ -1,677 +0,0 @@ -## Aurva Data Plane -## Ref: https://github.com/aurva-io/aurva-charts.git - -postgresql: - enabled: true - fullnameOverride: "aurva-dataplane-database" - volumePermissions: - ## @param volumePermissions.enabled Enable init container that changes the owner and group of the persistent volume - ## - enabled: true - global: - storageClass: hyperdisk-balanced - postgresql: - auth: - postgresPassword: "aurva" - database: "controller" - # Add toleration to make sure where this postgres db pod should reside (Applicable for production workloads): For more detail ref: https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/ - primary: - extendedConfiguration: | - max_connections = 300 - - #PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: farmiso-devops - - # -- Select nodes to deploy which matches the following labels - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: farmiso-devops - -# -- Provide a name in place of `aurva` -# namespaceOverride: aurva-dataplane - -########################################################## -# Global Configs -########################################################## -global: - aurva_controller: - enabled: true - aurva_fastdet: - enabled: false - aurva_pii_analyzer: - enabled: true - aurva_ocr: - enabled: true - aurva_collector: - enabled: true - - deploymentAnnotations: {} - - priorityClassName: "" - -########################################################## -# Aurva Controller -########################################################## -aurva_controller: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "farmiso" - team: "farmiso-devops" - service: "aurva-farmiso-a-prd" - env: "prd" - priority: "p0" - type: "aurva_controller" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 10 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - image: - # Image of the app container - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-controller/aurva-controller - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - - type: secret - name: aurva-controller-secrets - # - type: configmap - # name: proxy-datasource-config - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 2 - requests: - memory: 1Gi - cpu: 1 - - aurvaFastdet: - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-fastdet/aurva-fastdet - tag: "v2.30.13" - pullPolicy: IfNotPresent - envFrom: [] - env: [] - resources: - limits: - cpu: 0.5 - memory: 512Mi - requests: - cpu: 0.5 - memory: 512Mi - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: farmiso-devops - - # -- Taint tolerations for nodes - ##PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: farmiso-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-controller-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - config: - #variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - COMMAND_URL: "command.aurva-prd.meeshogcp.in:80" - DEPLOYMENT_TYPE: "kubernetes" - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - FLUSHER_WORKER_POOL_SIZE: "1000" - FLUSHER_BATCH_SIZE: "30000" - UNIQUENESS_IDENTIFIER: "gke-farmiso-prd-ase1a" #Recommendation: should be equal to cluster name - PROVIDER_ACCOUNT_ID: "meesho-farmiso-prd-0622" # GCP PROJECT ID (not Number) - REGION: "asia-southeast1" #eg: asia-south1 - ENVIRONMENT: "prod" - OCR_ENABLED: "true" - MONITORING_ENABLED: "false" - HYBRID_ONLY_MODE: "false" - FORCE_TLS: "false" - FASTDET_FLAG : "true" - AADHAAR_ENHANCER: "0" - WORKSPACE_EVENT_TRACKING_ENABLED: "false" - ACCESS_IQ_ENABLED: "true" - GRPC_ENFORCE_ALPN_ENABLED: "false" - ENABLE_FASTDET: "true" - FASTDET_MAX_BATCH_SIZE: "100" - HEARTBEAT_INTERVAL: "5m" - #pii data - PII_BUCKET_NAME: "gcs-infra-devop-aurva-farmiso-prd" - PII_LOG_BUCKET_REGION: "asia-southeast1" - PII_LOG_CRON: "*/5 * * * *" - ENABLE_PII_LOG: "true" - PII_EVIDENCE_UPLOAD_MAX_BLOCKING_TASKS: "50000" - PII_EVIDENCE_UPLOAD_MAX_CONCURRENT_TASKS: "500" - PII_EVIDENCE_MAX_CACHE_WEIGHT: "100" - QUOTA_CLEANUP_CRON: "0 0 * * *" - ENABLE_PII_QUOTA: "true" - MAX_PII_EVIDENCES_PER_KEY_PER_WINDOW: "3" - PII_QUOTA_SYNC_CRON: "*/2 * * * *" - QUOTA_WINDOW_HOURS: "24" - #constants - SKIP_NAMESPACES: "contour-internal-1-farmiso-a-prd,contour-internal-0-farmiso-a-prd-intra,contour-internal-0-farmiso-a-prd,contour-external-farmiso-a-prd" - CLOUD_PROVIDER: "gcp" - LOG_ENV: "production" - RDS_SCANNER_AVAILABILITY : "false" - REDSHIFT_SCANNER_AVAILABILITY : "false" - S3_SCANNER_AVAILABILITY : "false" - DYNAMO_SCANNER_AVAILABILITY: "false" - DOCDB_SCANNER_AVAILABILITY: "false" - OPENSEARCH_SCANNER_AVAILABILITY: "false" - CLOUDSQL_SCANNER_AVAILABILITY: "true" - BIGQUERY_SCANNER_AVAILABILITY: "true" - AWS_SNAPSHOT_SCANNER_AVAILABILITY: "false" - CLOUDSTORAGE_SCANNER_AVAILABILITY: "true" - KEYSPACES_SCANNER_AVAILABILITY: "false" - ALLOYDB_SCANNER_AVAILABILITY: "true" - BIGTABLE_SCANNER_AVAILABILITY: "true" - GCP_BACKUP_AVAILABILITY: "true" - EGRESS_MODE_ONLY: "false" - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - SCAN_UUID_ENABLED: "true" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-controller-sa - # -- Annotations applied to created service account - annotations: - iam.gke.io/gcp-service-account: sa-farmiso-prd-aurva-contr@meesho-farmiso-prd-0622.iam.gserviceaccount.com -# eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 15 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 - - -########################################################## -# Aurva OCR -########################################################## -aurva_ocr: - # -- Additional labels for aurva-controller - additionalLabels: - bu: "farmiso" - team: "farmiso-devops" - service: "aurva-farmiso-a-prd" - env: "prd" - priority: "p0" - type: "aurva_ocr" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 1 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-ocr/aurva-ocr - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-ocr: - type: secret - name: aurva-ocr-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 1 - requests: - memory: 2Gi - cpu: 1 - - # -- Select nodes to deploy which matches the following labels - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: farmiso-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: farmiso-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-ocr-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - OCR_TIME_LIMIT: "1" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "gke-farmiso-prd-ase1a" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-ocr-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # iam.gke.io/gcp-service-account: service-account@gcp.iam.gserviceaccount.com - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva Collector -########################################################## -aurva_collector: - - # -- Additional labels for aurva-analyzer - additionalLabels: - bu: "farmiso" - team: "farmiso-devops" - service: "aurva-farmiso-a-prd" - env: "prd" - priority: "p0" - type: "aurva_collector" - - - # -- Annotations on aurva-analyzer - annotations: {} -# "key": "value" - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-collector/aurva-collector - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-controller: - type: secret - name: aurva-collector-secrets - - resources: - limits: - cpu: 800m - memory: 800Mi - requests: - cpu: 200m - memory: 512Mi - - podSecurityContext: {} - - securityContext: - privileged: true - capabilities: - add: - # For kernel v5.8 and above we don't need CAP_SYS_ADMIN or CAP_SYS_RESOURCE - # we just need CAP_BPF and CAP_PERFMON. This has been tested on our EKS node - # which is on kernel v5.10.x - # When SSL Tracing is required we need CAP_SYS_ADMIN and CAP_SYS_PTRACE - # on top of the previous capabilities - # So finally these are the 4 possible combinations for capabilies - # 1. Newer Kernels without SSL - # - BPF - # - PERFMON - # 2. Newer Kernels with SSL - - SYS_ADMIN - - SYS_PTRACE - # 3. Older Kernels without SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # 4. Older Kernels with SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # - SYS_PTRACE - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - volumes: - - name: debugfs - mountPath: /sys/kernel/debug - hostPath: /sys/kernel/debug - - name: vmlinux - mountPath: /sys/kernel/btf/vmlinux - hostPath: /sys/kernel/btf/vmlinux - - name: procfs - mountPath: /host/proc - hostPath: /proc - - name: bpffs - mountPath: /sys/fs/bpf - hostPath: /sys/fs/bpf - - # -- Taint tolerations for nodes - tolerations: - # - effect: NoSchedule - # key: dedicated - # operator: Equal - # value: megatetra - - operator: Exists - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmagent-mds - - vminsert-mds - - vmselect-mds - - vmstorage-n4d - - contour-external-cc-v1 - - contour-internal-0-cc-v1 - - contour-intra-0-cc-v1 - - contour-shared-cc-v1 - - contour-shared - - alloy - - preprod-spot-16 - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-collector-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - # variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "gke-farmiso-prd-ase1a" - DEPLOYMENT_TYPE: "kubernetes" - TRACE_INTERNAL_SVC: "true" - TRACE_INTERNAL_SVC_HTTP: "true" - INTERNAL_SVC_SAMPLE_INTERVAL: "10m" - LOGS_TTL: "1h" - TRACE_HTTP2: "true" - TRACE_SSL: "false" - TRACE_PSQL: "false" - TRACE_SQLSERVER: "false" - TRACE_MYSQL: "false" - TRACE_EGRESS: "true" - TRACE_GO_TLS: "false" - TRACE_ML_SERVICES: "false" - # constants - LOG_ENV: production - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - MONITORING_ENABLED: "false" - ENABLE_INGRESS_INFORMER: "false" - ENABLE_SERVICE_INFORMER: "false" - ENABLE_ISTIO_INFORMER: "false" - EXCLUDED_PII_REGEX_TYPES: "ip_address,us_bank_number,us_driver_license,us_itin,us_passport,us_routing,us_mbi,ssn" - AGGREGATOR_MAX_CONNECTIONS: "1000" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-collector-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva PII Analyzer -########################################################## -aurva_pii_analyzer: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "farmiso" - team: "farmiso-devops" - service: "aurva-farmiso-a-prd" - env: "prd" - priority: "p0" - type: "aurva_pii_analyzer" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 3 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - ##PLACEHOLDER## - nodeSelector: - cloud.google.com/compute-class: farmiso-devops - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-piianalyzer/aurva-piianalyzer - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-pii-analyzer: - type: secret - name: aurva-pii-analyzer-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 4Gi - cpu: 4 - requests: - memory: 2Gi - cpu: 2 - - # nodeSelector: ##PLACEHOLDER## - # cloud.google.com/compute-class: farmiso-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: farmiso-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-pii-analyzer-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - SCHEDULER_TIME: "1" - SUPPORTED_REGION: "US" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "gke-farmiso-prd-ase1a" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-pii-analyzer-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 3 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 diff --git a/helm-overrides/gke-farmiso-prd-ase1a/cert-manager/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/cert-manager/custom-values.yaml deleted file mode 100644 index f8aaa9a..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/cert-manager/custom-values.yaml +++ /dev/null @@ -1,129 +0,0 @@ -global: - logLevel: 2 - rbac: - create: true - priorityClassName: "high-priority" -installCRDs: false - -crds: - enabled: true - keep: true - -# Cert-manager Controller -replicaCount: 1 -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-controller - tag: v1.20.2 - pullPolicy: IfNotPresent - -nodeSelector: - cloud.google.com/compute-class: farmiso-devops - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "farmiso-devops" - effect: "NoSchedule" - -resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 500m - memory: 512Mi - -# Webhook Configuration -webhook: - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-webhook - tag: v1.20.2 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: farmiso-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "farmiso-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# CA Injector Configuration -cainjector: - enabled: true - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-cainjector - tag: v1.20.2 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: farmiso-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "farmiso-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# ACME Solver Configuration -acmesolver: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-acmesolver - tag: v1.20.2 - pullPolicy: IfNotPresent - -# Startup API Check -startupapicheck: - enabled: true - timeout: 1m - backoffLimit: 4 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-startupapicheck - tag: v1.20.2 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: farmiso-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "farmiso-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - -# Prometheus Monitoring -prometheus: - enabled: true - servicemonitor: - enabled: false - interval: 60s - scrapeTimeout: 30s - labels: - prometheus: cert-manager diff --git a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/compacttetra-cc.yaml b/helm-overrides/gke-farmiso-prd-ase1a/computeclass/compacttetra-cc.yaml deleted file mode 100644 index f50cf1f..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/compacttetra-cc.yaml +++ /dev/null @@ -1,29 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: compacttetra -spec: - nodePoolConfig: - serviceAccount: sa-common-np-farm-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: compacttetra - team: farmiso-shared - taints: - - key: dedicated - value: compacttetra - effect: NoSchedule - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-standard-8 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-external-cc-v1.yaml b/helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-external-cc-v1.yaml deleted file mode 100644 index f4042de..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-external-cc-v1.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-external-cc-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-farm-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-internal-0-cc-v1.yaml b/helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-internal-0-cc-v1.yaml deleted file mode 100644 index 796b94a..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-internal-0-cc-v1.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-0-cc-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-farm-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-intra-0-cc-v1.yaml b/helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-intra-0-cc-v1.yaml deleted file mode 100644 index adebfe3..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-intra-0-cc-v1.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-0-cc-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-farm-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-shared-cc-v1.yaml b/helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-shared-cc-v1.yaml deleted file mode 100644 index ffb7d98..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-shared-cc-v1.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-shared-cc-v1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-farm-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-shared-cc.yaml b/helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-shared-cc.yaml deleted file mode 100644 index 2db24aa..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/contour-shared-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-shared -spec: - nodePoolConfig: - serviceAccount: sa-common-np-farm-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp \ No newline at end of file diff --git a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/farmiso-devops.yaml b/helm-overrides/gke-farmiso-prd-ase1a/computeclass/farmiso-devops.yaml deleted file mode 100644 index 47365b0..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/farmiso-devops.yaml +++ /dev/null @@ -1,34 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: farmiso-devops -spec: - nodePoolConfig: - serviceAccount: sa-common-np-farm-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-standard-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 60 - bootDiskType: hyperdisk-balanced - - machineType: c4-standard-4 - spot: false - maxPodsPerNode: 32 - storage: - bootDiskSize: 60 - bootDiskType: hyperdisk-balanced - - machineType: e2-standard-4 - spot: false - maxPodsPerNode: 32 - storage: - bootDiskSize: 60 - bootDiskType: pd-balanced - whenUnsatisfiable: DoNotScaleUp \ No newline at end of file diff --git a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/megaquad-cc.yaml b/helm-overrides/gke-farmiso-prd-ase1a/computeclass/megaquad-cc.yaml deleted file mode 100644 index 46f2ba9..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/megaquad-cc.yaml +++ /dev/null @@ -1,29 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaquad -spec: - nodePoolConfig: - serviceAccount: sa-common-np-farm-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megaquad - team: farmiso-shared - taints: - - key: dedicated - value: megaquad - effect: NoSchedule - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-standard-8 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/vmagent-mds-cc.yaml b/helm-overrides/gke-farmiso-prd-ase1a/computeclass/vmagent-mds-cc.yaml deleted file mode 100644 index 557b51a..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/vmagent-mds-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmagent-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-farm-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/vminsert-mds-cc.yaml b/helm-overrides/gke-farmiso-prd-ase1a/computeclass/vminsert-mds-cc.yaml deleted file mode 100644 index 1aef3ae..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/vminsert-mds-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vminsert-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-farm-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-highcpu-4 - maxPodsPerNode: 28 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/vmselect-mds-cc.yaml b/helm-overrides/gke-farmiso-prd-ase1a/computeclass/vmselect-mds-cc.yaml deleted file mode 100644 index 233a528..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/vmselect-mds-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmselect-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-farm-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml b/helm-overrides/gke-farmiso-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml deleted file mode 100644 index b10276e..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage-n4d -spec: - nodePoolConfig: - serviceAccount: sa-common-np-farm-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highmem-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-farmiso-prd-ase1a/conntrack-adjuster/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 02c4a82..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,17 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - max: 2097152 - hashsize: 524288 - sleepInterval: 30 - -nodeAffinity: - enabled: true - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: In - values: - - "contour-external" - - "contour-internal-0" diff --git a/helm-overrides/gke-farmiso-prd-ase1a/contour-cert-checker/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/contour-cert-checker/custom-values.yaml deleted file mode 100644 index 9204bfa..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=gke-farmiso-prd-ase1a"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - cloud.google.com/compute-class: farmiso-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: farmiso-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/gke-farmiso-prd-ase1a/contour-external/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/contour-external/custom-values.yaml deleted file mode 100644 index 8a26699..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/contour-external/custom-values.yaml +++ /dev/null @@ -1,115 +0,0 @@ -fullnameOverride: "contour-external-farmiso-a-prd" -configInline: - enableExternalNameService: true - network: - num-trusted-hops: 1 - timeouts: - connection-idle-timeout: 305s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 2 - podLabels: - bu: farmiso - team: farmiso-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-shared-cc-v1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: farmiso - team: farmiso-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-external-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-external-cc-v1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "40" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 6' - resources: - requests: - cpu: 6 - memory: 3Gi - limits: - cpu: 6 - memory: 12Gi - service: - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-farmiso-a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false - -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend-new" diff --git a/helm-overrides/gke-farmiso-prd-ase1a/contour-internal-0/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/contour-internal-0/custom-values.yaml deleted file mode 100644 index a489196..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,118 +0,0 @@ -fullnameOverride: "contour-internal-0-farmiso-a-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 2 - podLabels: - bu: farmiso - team: farmiso-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-shared-cc-v1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: farmiso - team: farmiso-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-0-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-internal-0-cc-v1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 2 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 6' - resources: - requests: - cpu: 6 - memory: 3Gi - limits: - cpu: 6 - memory: 12Gi - service: - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-farmiso-a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false - -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend-new" - diff --git a/helm-overrides/gke-farmiso-prd-ase1a/contour-internal-intra-0/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index ee4fdae..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,115 +0,0 @@ -fullnameOverride: "contour-internal-0-farmiso-intra-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 2 - podLabels: - bu: farmiso - team: farmiso-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-shared-cc-v1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: farmiso - team: farmiso-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-intra-0-cc-v1 - nodeSelector: - cloud.google.com/compute-class: contour-intra-0-cc-v1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 2 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 6' - resources: - requests: - cpu: 6 - memory: 3Gi - limits: - cpu: 6 - memory: 12Gi - service: - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false - -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend-new" diff --git a/helm-overrides/gke-farmiso-prd-ase1a/coredns/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/coredns/custom-values.yaml deleted file mode 100644 index 1ea907d..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/coredns/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -replicaCount: 5 -communicationType: "intra" - -labels: - bu: farmiso - team: farmiso-devops - env: prd - -clusterIP: 10.1.100.1 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "farmiso-devops" - effect: "NoSchedule" - -nodeSelector: - cloud.google.com/compute-class: farmiso-devops - kubernetes.io/os: linux \ No newline at end of file diff --git a/helm-overrides/gke-farmiso-prd-ase1a/external-secrets/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/external-secrets/custom-values.yaml deleted file mode 100644 index 616ecf3..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/external-secrets/custom-values.yaml +++ /dev/null @@ -1,61 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Annotations to add to Pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: farmiso-devops - nodeSelector: - cloud.google.com/compute-class: farmiso-devops - certController: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: farmiso-devops - nodeSelector: - cloud.google.com/compute-class: farmiso-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - webhook: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: farmiso-devops - nodeSelector: - cloud.google.com/compute-class: farmiso-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" \ No newline at end of file diff --git a/helm-overrides/gke-farmiso-prd-ase1a/flagger/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/flagger/custom-values.yaml deleted file mode 100644 index 79fec26..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/flagger/custom-values.yaml +++ /dev/null @@ -1,62 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "512Mi" - cpu: "1000m" - requests: - memory: "256Mi" - cpu: "100m" - -nodeSelector: - cloud.google.com/compute-class: farmiso-devops - -tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: farmiso-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: farmiso-devops - bu: infra - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-farmiso-rollout-service.prd-farmiso-rollout-service.svc.cluster.local" - -rolloutService: - enabled: true - apps: \ No newline at end of file diff --git a/helm-overrides/gke-farmiso-prd-ase1a/fluentd/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/fluentd/custom-values.yaml deleted file mode 100644 index 14c6ac2..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/fluentd/custom-values.yaml +++ /dev/null @@ -1,812 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-farm-fmsre-fluentd-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: farmiso - team: farmiso-sre - type: fluentd - service: fluentd-farmiso-a-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: farmiso - team: farmiso-sre - type: fluentd - service: fluentd-farmiso-a-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend-new - path: meesho/prd/cntr/devop/coralogix-keys - -envFrom: -# - secretRef: -# name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: - - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - from_encoding "utf-8" - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/gke-farmiso-prd-ase1a/keda/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/keda/custom-values.yaml deleted file mode 100644 index 57b58f4..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/keda/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - cloud.google.com/compute-class: farmiso-devops - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "farmiso-devops" - effect: "NoSchedule" - podLabels: - bu: "farmiso" - team: "farmiso-devops" - metricsAdapter: - bu: "farmiso" - team: "farmiso-devops" - resources: - webhooks: - limits: - cpu: 50m - memory: 100Mi - requests: - cpu: 10m - memory: 20Mi \ No newline at end of file diff --git a/helm-overrides/gke-farmiso-prd-ase1a/kube-dns/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/kube-dns/custom-values.yaml deleted file mode 100644 index e16f4cb..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.1.100.1"],"prd.mrouter.int.svc.cluster.local":["10.1.100.1"],"mq-server.meeshoint.in.svc.cluster.local":["10.1.100.1"]} diff --git a/helm-overrides/gke-farmiso-prd-ase1a/kube-events/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/kube-events/custom-values.yaml deleted file mode 100644 index 39baba4..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-farmiso-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - cloud.google.com/compute-class: farmiso-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: farmiso-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: farmiso-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: farmiso-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: farmiso-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: farmiso-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/gke-farmiso-prd-ase1a/kube-state-metrics/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 73c195d..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,427 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: "" - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-farmiso-a-prd - -imagePullSecrets: [] -# - name: "image-pull-secret" - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "kube-state-metrics-farmiso-a-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - cloud.google.com/compute-class: farmiso-devops -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: farmiso-devops - operator: Equal -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 100m - memory: 100Mi - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/gke-farmiso-prd-ase1a/kubectl-mcp-server/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/kubectl-mcp-server/custom-values.yaml deleted file mode 100644 index cda9aff..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/kubectl-mcp-server/custom-values.yaml +++ /dev/null @@ -1,119 +0,0 @@ -fullnameOverride: "kubectl-mcp-server" - -replicas: 1 - -image: - # In-house hardened build (helm-templates/kubectl-mcp-server/docker/) — - # NOT the upstream Docker Hub image. 266→18 HIGH/CRIT vulns, non-root, - # multi-stage slim base, kubectl v1.33.12 / helm v3.21.0. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/prd/devop/kubectl-mcp-server - tag: "v2" - pullPolicy: IfNotPresent - -labels: - bu: infra - team: devops - service: kubectl-mcp-server - env: prd - -serviceAccount: - create: true - annotations: {} - -rbac: - create: true - -podAnnotations: {} - -podSecurityContext: {} - -securityContext: - runAsNonRoot: true - runAsUser: 1000 - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -priorityClassName: "" - -nodeSelector: - cloud.google.com/compute-class: "farmiso-devops" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "farmiso-devops" - effect: NoSchedule - -affinity: {} - -resources: - requests: - cpu: 250m - memory: 512Mi - limits: - cpu: 1000m - memory: 1Gi - -mcp: - mode: single - transport: http - host: "0.0.0.0" - port: 8000 - -auth: - allowAnonymous: false - -externalSecrets: - enabled: true - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/prd/cntr/devop/kubectl-mcp-server-farmiso" - -# tcpSocket on purpose — streamable-http transport exposes only /mcp, -# there is no /health route (see chart values.yaml note). -livenessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - -readinessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - -service: - port: 8000 - type: ClusterIP - -# Contour standard (2.0.0-style): renders HTTPProxy parent/child + -intra -# instead of a networking.k8s.io Ingress. -createContourGateway: true - -ingress: - enabled: true - # farmiso prd exposes the -0 contour internal classes (no -1); the chart - # auto-derives contour-internal-intra-0. Verified on k8s-farmiso-prd-ase1. - ingressClassName: contour-internal-0 - servicePortNumber: 8000 - hosts: - - host: kubectl-mcp-server-farmiso.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 diff --git a/helm-overrides/gke-farmiso-prd-ase1a/kyverno/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/kyverno/custom-values.yaml deleted file mode 100644 index b91bcb2..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/kyverno/custom-values.yaml +++ /dev/null @@ -1,2243 +0,0 @@ -global: - - # -- Internal settings used with `helm template` to generate install manifest - # @ignored - templating: - enabled: false - debug: false - version: ~ - - image: - # -- (string) Global value that allows to set a single image registry across all deployments. - # When set, it will override any values set under `.image.registry` across the chart. - registry: ~ - # -- (list) Global list of Image pull secrets - # When set, it will override any values set under `imagePullSecrets` under different components across the chart. - imagePullSecrets: [] - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - caCertificates: - # -- Global CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - # Individual controller values will override this global value - data: ~ - - # -- Global value to set single volume to be mounted for CA certificates for all deployments. - # Not used when `.Values.global.caCertificates.data` is defined - # Individual controller values will override this global value - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Additional container environment variables to apply to all containers and init containers - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Global node labels for pod assignment. Non-global values will override the global value. - nodeSelector: - dedicated: farmiso-kyverno - - # -- Global List of node taints to tolerate. Non-global values will override the global value. - tolerations: - - key: "dedicated" - operator: "Equal" - value: "farmiso-kyverno" - effect: "NoSchedule" - -# -- (string) Override the name of the chart -nameOverride: ~ - -# -- (string) Override the expanded name of the chart -fullnameOverride: ~ - -# -- (string) Override the namespace the chart deploys to -namespaceOverride: ~ - -upgrade: - # -- Upgrading from v2 to v3 is not allowed by default, set this to true once changes have been reviewed. - fromV2: false - -apiVersionOverride: - # -- (string) Override api version used to create `PodDisruptionBudget`` resources. - # When not specified the chart will check if `policy/v1/PodDisruptionBudget` is available to - # determine the api version automatically. - podDisruptionBudget: ~ - -rbac: - roles: - # -- Aggregate ClusterRoles to Kubernetes default user-facing roles. For more information, see [User-facing roles](https://kubernetes.io/docs/reference/access-authn-authz/rbac/#user-facing-roles) - aggregate: - admin: true - view: true - -# Use openreports.io as the API group for reporting -openreports: - # -- Enable OpenReports feature in controllers - enabled: false - # -- Whether to install CRDs from the upstream OpenReports chart. Setting this to true requires enabled to also be true. - installCrds: false - -# CRDs configuration -crds: - - # -- Whether to have Helm install the Kyverno CRDs, if the CRDs are not installed by Helm, they must be added before policies can be created - install: true - - reportsServer: - # -- Kyverno reports-server is used in your cluster - enabled: false - - groups: - - # -- Install CRDs in group `kyverno.io` - kyverno: - cleanuppolicies: true - clustercleanuppolicies: true - clusterpolicies: true - globalcontextentries: true - policies: true - policyexceptions: true - updaterequests: true - - # -- Install CRDs in group `policies.kyverno.io` - policies: - validatingpolicies: true - policyexceptions: true - imagevalidatingpolicies: true - namespacedimagevalidatingpolicies: true - mutatingpolicies: true - generatingpolicies: true - deletingpolicies: true - namespaceddeletingpolicies: true - namespacedvalidatingpolicies: true - - # -- Install CRDs in group `reports.kyverno.io` - reports: - clusterephemeralreports: true - ephemeralreports: true - - # -- Install CRDs in group `wgpolicyk8s.io` - wgpolicyk8s: - clusterpolicyreports: false ##lk - policyreports: false ##lk - - # -- Additional CRDs annotations (Replace=true avoids 262144-byte annotation limit on large CRDs) - annotations: - argocd.argoproj.io/sync-options: Replace=true - # strategy.spinnaker.io/replace: 'true' - - # -- Additional CRDs labels - customLabels: {} - - migration: - - # -- Enable CRDs migration using helm post upgrade hook - enabled: true - - # -- Resources to migrate - resources: - - cleanuppolicies.kyverno.io - - clustercleanuppolicies.kyverno.io - - clusterpolicies.kyverno.io - - globalcontextentries.kyverno.io - - policies.kyverno.io - - policyexceptions.kyverno.io - - updaterequests.kyverno.io - - deletingpolicies.policies.kyverno.io - - generatingpolicies.policies.kyverno.io - - imagevalidatingpolicies.policies.kyverno.io - - namespacedimagevalidatingpolicies.policies.kyverno.io - - mutatingpolicies.policies.kyverno.io - - namespaceddeletingpolicies.policies.kyverno.io - - namespacedvalidatingpolicies.policies.kyverno.io - - policyexceptions.policies.kyverno.io - - validatingpolicies.policies.kyverno.io - - image: - # -- (string) Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- (string) Image repository - repository: kyverno/kyverno-cli - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- (string) Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podResources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -# Configuration -config: - - # -- Create the configmap. - create: true - - # -- Preserve the configmap settings during upgrade. - preserve: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - # -- Enable registry mutation for container images. Enabled by default. - enableDefaultRegistryMutation: true - - # -- The registry hostname used for the image mutation. - defaultRegistry: docker.io - - # -- Exclude groups - excludeGroups: - - system:nodes - - # -- Exclude usernames - excludeUsernames: [] - # - '!system:kube-scheduler' - - # -- Exclude roles - excludeRoles: [] - - # -- Exclude roles - excludeClusterRoles: [] - - # -- Generate success events. - generateSuccessEvents: false - - # -- Resource types to be skipped by the Kyverno policy engine. - # Make sure to surround each entry in quotes so that it doesn't get parsed as a nested YAML list. - # These are joined together without spaces, run through `tpl`, and the result is set in the config map. - # @default -- See [values.yaml](values.yaml) - resourceFilters: - - '[Event,*,*]' - # - '[*/*,kube-system,*]' - - '[*/*,kube-public,*]' - - '[*/*,kube-node-lease,*]' - - '[Node,*,*]' - - '[Node/?*,*,*]' - - '[APIService,*,*]' - - '[APIService/?*,*,*]' - - '[TokenReview,*,*]' - - '[SubjectAccessReview,*,*]' - - '[SelfSubjectAccessReview,*,*]' - - '[Binding,*,*]' - - '[Pod/binding,*,*]' - - '[ReplicaSet,*,*]' - - '[ReplicaSet/?*,*,*]' - - '[EphemeralReport,*,*]' - - '[ClusterEphemeralReport,*,*]' - # exclude resources from the chart - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:additional]' - - '[ClusterRoleBinding,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.configMapName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.metricsConfigMapName" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Job,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[Job/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.admission-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.background-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - # -- Sets the threshold for the total number of UpdateRequests generated for mutateExisitng and generate policies. - updateRequestThreshold: 1000 - - # -- Defines the `namespaceSelector`/`objectSelector` in the webhook configurations. - # The Kyverno namespace is excluded if `excludeKyvernoNamespace` is `true` (default) - webhooks: - # Exclude namespaces - # namespaceSelector: lk-s2 - # matchExpressions: - # - key: kubernetes.io/metadata.name - # operator: NotIn - # values: - # - kube-systemm lk-s2 - # Exclude objects - # objectSelector: - # matchExpressions: - # - key: webhooks.kyverno.io/exclude - # operator: DoesNotExist - - # -- Defines annotations to set on webhook configurations. - webhookAnnotations: - # Example to disable admission enforcer on AKS: - # 'admissions.enforcer/disabled': 'true' lk-s3 - - # -- Defines labels to set on webhook configurations. - webhookLabels: {} - # Example to adopt webhook resources in ArgoCD: - # 'argocd.argoproj.io/instance': 'kyverno' - - # -- Defines match conditions to set on webhook configurations (requires Kubernetes 1.27+). - matchConditions: [] - - # -- Exclude Kyverno namespace - # Determines if default Kyverno namespace exclusion is enabled for webhooks and resourceFilters - excludeKyvernoNamespace: true - - # -- resourceFilter namespace exclude - # Namespaces to exclude from the default resourceFilters - resourceFiltersExcludeNamespaces: [] - - # -- resourceFilters exclude list - # Items to exclude from config.resourceFilters - resourceFiltersExclude: [] - - # -- resourceFilter namespace include - # Namespaces to include to the default resourceFilters - resourceFiltersIncludeNamespaces: [] - - # -- resourceFilters include list - # Items to include to config.resourceFilters - resourceFiltersInclude: [] - -# Metrics configuration -metricsConfig: - - # -- Create the configmap. - create: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - namespaces: - - # -- List of namespaces to capture metrics for. - include: [] - - # -- list of namespaces to NOT capture metrics for. - exclude: [] - - # -- (string) Rate at which metrics should reset so as to clean up the memory footprint of kyverno metrics, if you might be expecting high memory footprint of Kyverno's metrics. Default: 0, no refresh of metrics. WARNING: This flag is not working since Kyverno 1.8.0 - metricsRefreshInterval: ~ - # metricsRefreshInterval: 24h - - # -- (list) Configures the bucket boundaries for all Histogram metrics, changing this configuration requires restart of the kyverno admission controller - bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 15, 20, 25, 30] - - # -- (map) Configures the exposure of individual metrics, by default all metrics and all labels are exported, changing this configuration requires restart of the kyverno admission controller - metricsExposure: - kyverno_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_image_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_mutating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_generating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_admission_review_duration_seconds: - # enabled: false - disabledLabelDimensions: ["resource_namespace"] - kyverno_policy_rule_info_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_policy_results_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_admission_requests_total: - disabledLabelDimensions: ["resource_namespace"] - kyverno_cleanup_controller_deletedobjects_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - -# -- Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -imagePullSecrets: {} - # regcred: - # registry: foo.example.com - # username: foobar - # password: secret - # regcred2: - # registry: bar.example.com - # username: barbaz - # password: secret2 - -# -- Existing Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -existingImagePullSecrets: [] - # - test-registry - # - other-test-registry - -# Tests configuration -test: - # -- Sleep time before running test - sleep: 20 - - image: - # -- (string) Image registry - registry: curlimages - # -- Image repository - repository: curl - # -- Image tag - # Defaults to `latest` if omitted - tag: '8.10.1' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - # -- Security context for the test containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- Additional Pod annotations - podAnnotations: {} - - # -- List of node taints to tolerate - tolerations: [] - -# -- Additional labels -customLabels: {} - - -webhooksCleanup: - # -- Create a helm pre-delete hook to cleanup webhooks. - enabled: true - - autoDeleteWebhooks: - # -- Allow webhooks controller to delete webhooks using finalizers - enabled: false - - image: - # -- (string) Image registry - registry: registry.k8s.io - # -- Image repository - repository: kubectl - # -- Image tag - # Defaults to `latest` if omitted - tag: 'v1.32.7' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -grafana: - # -- Enable grafana dashboard creation. - enabled: false - - # -- Configmap name template. - configMapName: '{{ include "kyverno.fullname" . }}-grafana' - - # -- (string) Namespace to create the grafana dashboard configmap. - # If not set, it will be created in the same namespace where the chart is deployed. - namespace: ~ - - # -- Grafana dashboard configmap annotations. - annotations: {} - - # -- Grafana dashboard configmap labels - labels: - grafana_dashboard: "1" - - # -- create GrafanaDashboard custom resource referencing to the configMap. - # according to https://grafana-operator.github.io/grafana-operator/docs/examples/dashboard_from_configmap/readme/ - grafanaDashboard: - create: false - folder: kyverno - allowCrossNamespaceImport: true - matchLabels: - dashboards: "grafana" - -# Features configuration -features: - admissionReports: - # -- Enables the feature - enabled: true - aggregateReports: - # -- Enables the feature - enabled: true - policyReports: - # -- Enables the feature - enabled: false ##lk - validatingAdmissionPolicyReports: - # -- Enables the feature - enabled: true - mutatingAdmissionPolicyReports: - # -- Enables the feature - enabled: false - reporting: - # -- Enables the feature - validate: true - # -- Enables the feature - mutate: true - # -- Enables the feature - mutateExisting: true - # -- Enables the feature - imageVerify: true - # -- Enables the feature - generate: true - autoUpdateWebhooks: - # -- Enables the feature - enabled: true - backgroundScan: - # -- Enables the feature - enabled: true - # -- Number of background scan workers - backgroundScanWorkers: 2 - # -- Background scan interval - backgroundScanInterval: 1h - # -- Skips resource filters in background scan - skipResourceFilters: true - configMapCaching: - # -- Enables the feature - enabled: true - controllerRuntimeMetrics: - # -- Bind address for controller-runtime metrics (use "0" to disable it) - bindAddress: ":8080" - deferredLoading: - # -- Enables the feature - enabled: true - dumpPayload: - # -- Enables the feature - enabled: false - forceFailurePolicyIgnore: - # -- Enables the feature - enabled: false - generateValidatingAdmissionPolicy: - # -- Enables the feature - enabled: true - generateMutatingAdmissionPolicy: - # -- Enables the feature - enabled: false - dumpPatches: - # -- Enables the feature - enabled: false - globalContext: - # -- Maximum allowed response size from API Calls. A value of 0 bypasses checks (not recommended) - maxApiCallResponseLength: 2000000 - logging: - # -- Logging format - format: text - # -- Logging verbosity - verbosity: 2 - omitEvents: - # -- Events which should not be emitted (possible values `PolicyViolation`, `PolicyApplied`, `PolicyError`, and `PolicySkipped`) - eventTypes: - - PolicyApplied - - PolicySkipped - # - PolicyViolation - # - PolicyError - policyExceptions: - # -- Enables the feature - enabled: true - # -- Restrict policy exceptions to a single namespace - # Set to "*" to allow exceptions in all namespaces - namespace: '' - protectManagedResources: - # -- Enables the feature - enabled: false - registryClient: - # -- Allow insecure registry - allowInsecure: false - # -- Enable registry client helpers - credentialHelpers: - - default - - google - - amazon - - azure - - github - ttlController: - # -- Reconciliation interval for the label based cleanup manager - reconciliationInterval: 1m - tuf: - # -- Enables the feature - enabled: false - # -- (string) Path to Tuf root - root: ~ - # -- (string) Raw Tuf root - rootRaw: ~ - # -- (string) Tuf mirror - mirror: ~ - -# Admission controller configuration -admissionController: - autoscaling: - # -- Enable horizontal pod autoscaling - enabled: true - - # -- Minimum number of pods - minReplicas: 1 - - # -- Maximum number of pods - maxReplicas: 10 - - # -- Target CPU utilization percentage - targetCPUUtilizationPercentage: 80 - - # -- Configurable scaling behavior - behavior: {} - - # -- Overrides features defined at the root level - featuresOverride: - admissionReports: - # -- Max number of admission reports allowed in flight until the admission controller stops creating new ones - backPressureThreshold: 1000 - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- The ServiceAccount name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno admission controller activities. - # This will help ensure Kyverno stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- admissionController webhook server port - # in case you are using hostNetwork: true, you might want to change the port the webhookServer is listening to - webhookServer: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - admission-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.admissionController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - initContainer: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyvernopre - # -- (string) Image tag - # If missing, defaults to image.tag - tag: ~ - # -- (string) Image pull policy - # If missing, defaults to image.pullPolicy - pullPolicy: ~ - - resources: - # -- Pod resource limits - limits: - cpu: 1000m - memory: 1Gi - # -- Pod resource requests - requests: - cpu: 1000m - memory: 1Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - container: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyverno - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - resources: - # -- Pod resource limits - limits: - cpu: 1900m - memory: 3Gi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Array of extra init containers - extraInitContainers: [] - # - name: init-container - # image: busybox - # command: ['sh', '-c', 'echo Hello'] - - # -- Array of extra containers to run alongside kyverno - extraContainers: [] - # - name: myapp-container - # image: busybox - # command: ['sh', '-c', 'echo Hello && sleep 3600'] - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Kyverno's metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Background controller configuration -backgroundController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable background controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: - - apiGroups: - - networking.k8s.io - resources: - - ingresses - - ingressclasses - - networkpolicies - verbs: - - create - - update - - patch - - delete - - apiGroups: - - rbac.authorization.k8s.io - resources: - - rolebindings - - roles - verbs: - - create - - update - - patch - - delete - - apiGroups: - - '' - resources: - - configmaps - - resourcequotas - - limitranges - verbs: - - create - - update - - patch - - delete - - apiGroups: - - resource.k8s.io - resources: - - resourceclaims - - resourceclaimtemplates - verbs: - - create - - delete - - update - - patch - - deletecollection - - apiGroups: - - autoscaling - resources: - - horizontalpodautoscalers - verbs: - - get - - list - - watch - - update - - patch - - apiGroups: - - apps - resources: - - deployments - - statefulsets - - replicasets - verbs: - - get - - list - - watch - - update - - patch - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - # - patch - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/background-controller - # -- Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 1900m - memory: 2Gi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - background-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.backgroundController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - # -- backgroundController server port - # in case you are using hostNetwork: true, you might want to change the port the backgroundController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Cleanup controller configuration -cleanupController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable cleanup controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - delete - # - list - # - watch - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/cleanup-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- cleanupController server port - # in case you are using hostNetwork: true, you might want to change the port the cleanupController is listening to - server: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 1100m - memory: 1.2Gi - # -- Pod resource requests - requests: - cpu: 1100m - memory: 1.2Gi - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - cleanup-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `service.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Reports controller configuration -reportsController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable reports controller. - enabled: false ##lk - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/reports-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno reports controller activities. - # This will help ensure Kyverno reports stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - memory: 128Mi - # -- Pod resource requests - requests: - cpu: 100m - memory: 64Mi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - reports-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.reportsController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- (string) Service node port. - # Only used if `type` is `NodePort`. - nodePort: ~ - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- (string) Traces receiver address - address: ~ - # -- (string) Traces receiver port - port: ~ - # -- (string) Traces receiver credentials - creds: ~ - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- (string) Otel collector endpoint - collector: ~ - # -- (string) Otel collector credentials - creds: ~ - - # -- reportsController server port - # in case you are using hostNetwork: true, you might want to change the port the reportsController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - - # -- Enable sanity check for reports CRDs - sanityChecks: true diff --git a/helm-overrides/gke-farmiso-prd-ase1a/kyverno/policies/protect-namespace.yaml b/helm-overrides/gke-farmiso-prd-ase1a/kyverno/policies/protect-namespace.yaml deleted file mode 100644 index 1061f7c..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/kyverno/policies/protect-namespace.yaml +++ /dev/null @@ -1,40 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: protect-critical-namespaces -spec: - background: false - rules: - - name: deny-deletion-of-critical-namespaces - match: - any: - - resources: - kinds: - - Namespace - names: - - keda-farmiso-prd - - contour-internal-0-farmiso-prd - - contour-internal-0-farmiso-prd-intra - - contour-external-farmiso-prd - - external-secrets-farmiso-prd - - flagger-farmiso-prd - - victoriametrics - - kube-system - - monitoring - - telegraf-operator - - loadtester - - opentelemetry - - fluentd - - kube-events - - observability - - contour-cert-checker-ns - - aurva-dataplane - validate: - failureAction: Enforce - message: "Deletion of critical namespaces is not allowed." - deny: - conditions: - any: - - key: "{{request.operation}}" - operator: Equals - value: DELETE diff --git a/helm-overrides/gke-farmiso-prd-ase1a/kyverno/policies/restrict-replicas.yaml b/helm-overrides/gke-farmiso-prd-ase1a/kyverno/policies/restrict-replicas.yaml deleted file mode 100644 index 101bbeb..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/kyverno/policies/restrict-replicas.yaml +++ /dev/null @@ -1,29 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: restrict-minimum-replicas-selected-ns -spec: - background: false - rules: - - name: min-replicas-deployments-statefulsets - match: - resources: - kinds: - - Deployment - - StatefulSet - - Deployment/scale - - StatefulSet/scale - names: - - coredns - namespaces: - - kube-system - operations: - - CREATE - - UPDATE - validate: - failureAction: Enforce - message: "Deployments and StatefulSets must have at least 5 replicas in this namespace." - anyPattern: - # Case 1: replicas explicitly set and >= 5 - - spec: - replicas: ">=5" diff --git a/helm-overrides/gke-farmiso-prd-ase1a/loadtester/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/loadtester/custom-values.yaml deleted file mode 100644 index 6444246..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/loadtester/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -replicaCount: 1 - -Namespace: loadtester - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/dev/devops/flagger-loadtester - tag: 0.30.0 - pullPolicy: IfNotPresent - pullSecret: - -podLabels: - bu: farmiso - team: farmiso-shared - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - -podPriorityClassName: "" - -logLevel: info -cmd: - timeout: 1h - namespaceRegexp: "" - -nameOverride: "loadtester" -fullnameOverride: "" - -env: [] - -service: - type: ClusterIP - port: 80 - -resources: - requests: - cpu: 10m - memory: 64Mi - -volumes: [] -volumeMounts: [] - -nodeSelector: - cloud.google.com/compute-class: farmiso-devops - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "farmiso-devops" - effect: "NoSchedule" - -affinity: {} - -ingress: - slowStart: - enabled: false - enabled: true - ingressClassName: contour-internal-0 - servicePort: http - hosts: - - host: farmiso-flagger-loadtester.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / -createContourGateway: true -contourResponseTimeout: false -rbac: - # rbac.create: `true` if rbac resources should be created - create: true - # rbac.scope: `cluster` to create cluster-scope rbac resources (ClusterRole/ClusterRoleBinding) - # otherwise, namespace-scope rbac resources will be created (Role/RoleBinding) - scope: - # rbac.rules: array of rules to apply to the role. example: - # rules: - # - apiGroups: [""] - # resources: ["pods"] - # verbs: ["list", "get"] - rules: [] - -# name of an existing service account to use - if not creating rbac resources -serviceAccountName: "" - -# App Mesh virtual node settings (to be used for AppMesh v1beta1) -meshName: "" -#backends: -# - app1.namespace -# - app2.namespace - -# App Mesh virtual node settings (to be used for AppMesh v1beta2) -appmesh: - enabled: false - backends: - - podinfo - - podinfo-canary - -#Istio virtual service and gatway settings. TLS secrets should be in namespace before enbaled it. ( secret format loadtester.fullname ) -istio: - enabled: false - host: flagger-loadtester.flagger - gateway: - enabled: false - tls: - enabled: false - httpsRedirect: false - -# when enabled, it will add a security context for the loadtester pod -securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 100 - runAsGroup: 101 - -podDisruptionBudget: - enabled: false - minAvailable: 1 diff --git a/helm-overrides/gke-farmiso-prd-ase1a/opentelemetry-coralogix/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/opentelemetry-coralogix/custom-values.yaml deleted file mode 100644 index cf6203e..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/opentelemetry-coralogix/custom-values.yaml +++ /dev/null @@ -1,281 +0,0 @@ -global: - domain: "coralogixsg.com" - defaultApplicationName: "default" - defaultSubsystemName: "nodes" - - # Old endpoint based configuration, - # please use domain instead. - traces: - endpoint: "" - metrics: - endpoint: "" - logs: - endpoint: "" - -# set distribution to openshift for openshift clusters -distribution: "" - -# Opentelemetry collector configuration -dedicatedValue: false -schedulerName: default-scheduler - -labels: - bu: "farmiso" - team: "farmiso-sre" - service: "opentelemetry-farmiso-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: false - key: prd/admin/coralogix-keys - secretStoreRef: - name: vault-backend-new - -image: - # If you want to use the core image `otel/opentelemetry-collector`, you also need to change `command.name` value to `otelcol`. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - # repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - tag: "0.153.0" - # When digest is set to a non-empty value, images will be pulled by digest (regardless of tag value). - digest: "" -mode: daemonset -rollout: - rollingUpdate: - # When 'mode: daemonset', maxSurge cannot be used when hostPort is set for any of the ports - # maxSurge: 25% - maxUnavailable: 1 - strategy: RollingUpdate -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" -hostNetwork: true -dnsPolicy: "ClusterFirstWithHostNet" -fullnameOverride: opentelemetry-farmiso-prd - -presets: - logsCollection: - enabled: false - storeCheckpoints: false - maxRecombineLogSize: 1048576 - extraFilelogOperators: [] -# - type: recombine -# combine_field: body -# source_identifier: attributes["log.file.path"] -# is_first_entry: body matches "^(YOUR-LOGS-REGEX)"s - kubernetesAttributes: - enabled: false - hostMetrics: - enabled: false - kubeletMetrics: - enabled: false - -extraEnvs: -- name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" -- name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName -config: - extensions: {} - # zpages: - # endpoint: localhost:55679 - # pprof: - # endpoint: localhost:1777 - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 4s - tls: - insecure: true - sending_queue: - enabled: true - queue_size: 6000 - num_consumers: 200 - resolver: - dns: - hostname: opentelemetry-admin-prd.opentelemetry.svc.clusterset.local - # coralogix: - # timeout: "30s" - # private_key: "${CORALOGIX_PRIVATE_KEY}" - # domain: "{{.Values.global.domain}}" - # traces: - # endpoint: "{{ .Values.global.traces.endpoint }}" - # metrics: - # endpoint: "{{ .Values.global.metrics.endpoint }}" - # logs: - # endpoint: "{{ .Values.global.logs.endpoint }}" - # application_name_attributes: - # - "k8s.namespace.name" - # - "service.namespace" - # subsystem_name_attributes: - # - "k8s.deployment.name" - # - "k8s.statefulset.name" - # - "k8s.daemonset.name" - # - "k8s.cronjob.name" - # - "k8s.job.name" - # - "k8s.container.name" - # - "k8s.node.name" - # - "service.name" - # application_name: "{{.Values.global.defaultApplicationName }}" - # subsystem_name: "{{.Values.global.defaultSubsystemName }}" - processors: - k8s_attributes: - filter: - node_from_env_var: KUBE_NODE_NAME - extract: - metadata: - - "k8s.namespace.name" - - "k8s.deployment.name" - - "k8s.statefulset.name" - - "k8s.daemonset.name" - - "k8s.cronjob.name" - - "k8s.job.name" - - "k8s.pod.name" - - "k8s.node.name" - memory_limiter: null # Will get the k8s resource limits - resourcedetection/env: - detectors: ["system","env"] - timeout: 2s - override: false - # spanmetrics: - # metrics_exporter: coralogix - # dimensions: - # - name: "k8s.deployment.name" - # - name: "k8s.statefulset.name" - # - name: "k8s.daemonset.name" - # - name: "k8s.cronjob.name" - # - name: "k8s.job.name" - # - name: "k8s.container.name" - # - name: "k8s.node.name" - # - name: "k8s.namespace.name" - receivers: - otlp: - protocols: - grpc: - endpoint: ${MY_POD_IP}:4317 - http: - endpoint: ${MY_POD_IP}:4318 - zipkin: - endpoint: ${MY_POD_IP}:9411 - jaeger: - protocols: - grpc: - endpoint: ${MY_POD_IP}:14250 - thrift_http: - endpoint: ${MY_POD_IP}:14268 - thrift_compact: - endpoint: ${MY_POD_IP}:6831 - thrift_binary: - endpoint: ${MY_POD_IP}:6832 - prometheus: - config: - scrape_configs: - - job_name: opentelemetry-collector - scrape_interval: 30s - static_configs: - - targets: - - ${MY_POD_IP}:8888 - service: - extensions: - # - zpages - # - pprof - - health_check -# - memory_ballast - telemetry: - metrics: - address: ${env:MY_POD_IP}:8888 - pipelines: - traces: - exporters: - - loadbalancing - processors: - - memory_limiter - # - spanmetrics - - batch - receivers: - - otlp - metrics: null - logs: null -tolerations: - - operator: Exists - -resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 1 - memory: 2G - -ports: - jaeger-binary: - enabled: true - containerPort: 6832 - servicePort: 6832 - hostPort: 6832 - protocol: TCP - # In order to enable podMonitor, following part must be enabled in order to expose the required port: - # metrics: - # enabled: true - -# podMonitor: -# enabled: true - -# prometheusRule: -# enabled: true -# defaultRules: -# enabled: true - -# Annotations to be added to pod -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-supply-contour-ext - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-farmiso-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert \ No newline at end of file diff --git a/helm-overrides/gke-farmiso-prd-ase1a/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index e6820a0..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,134 +0,0 @@ -fullnameOverride: opentelemetry-farmiso-prd - -mode: daemonset - -priorityClassName: "system-node-critical" - -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.153.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-supply-contour-ext - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-farmiso-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external - -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 100m - memory: 124Mi - limits: - cpu: 200m - memory: 256Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "farmiso" - team: "sre" - service: "opentelemetry-farmiso-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/gke-farmiso-prd-ase1a/prometheus-node-exporter/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index cdb7cf1..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,487 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-farmiso-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "node-exporter-farmiso-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/gke-farmiso-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index c60d736..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,169 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-farmiso-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-farmiso-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "stackdriver-exporter-farmiso-prd" - env: "prd" - priority: "p0" - type: "exporter" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-farmiso-prd-0622" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'compute.googleapis.com/instance,cloudsql.googleapis.com/database,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - cloud.google.com/compute-class: farmiso-devops - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - effect: "NoSchedule" - key: "cloud.google.com/compute-class" - value: "farmiso-devops" - operator: "Equal" - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-stackdriver-exp-farmiso-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - } - - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/gke-farmiso-prd-ase1a/telegraf-operator/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/telegraf-operator/custom-values.yaml deleted file mode 100644 index 7a85ace..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,151 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "farmiso" - team: "farmiso-sre" - service: "telegraf-operator-farmiso-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - cloud.google.com/compute-class: "farmiso-devops" - -tolerations: - - effect: "NoSchedule" - key: "cloud.google.com/compute-class" - value: "farmiso-devops" - operator: "Equal" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false diff --git a/helm-overrides/gke-farmiso-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml deleted file mode 100644 index fc08e18..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml +++ /dev/null @@ -1,296 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-farmiso-prd-dr -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-farmiso-prd-dr-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent-dr - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-farm-fmsre-vmagent-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - - https://vminsert-infra-prd-dr.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "vmagent-farmiso-prd" - env: "prd" - priority: "p0" - type: "vmagent-dr" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-farmiso-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-farmiso-prd-dr.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 3 - memory: 2Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent-dr" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/gke-farmiso-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml deleted file mode 100644 index fb1a48e..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml +++ /dev/null @@ -1,272 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 1 - -fullnameOverride: vmagent-farmiso-a-prd-fb -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-farmiso-a-prd-fb-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - -# Remove topology spread constraints -topologySpreadConstraints: [] - -# Modify deployment strategy -deployment: - enabled: true - -# Remove HPA -horizontalPodAutoscaler: - enabled: false - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-infr-sre-vmagent-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -# Remove pod disruption budget -podDisruptionBudget: - enabled: false - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-farmiso.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vmsinglenode-infra-prd.victoriametrics.svc.clusterset.local:8428/api/v1/write - - https://vmsinglenode-infra-prd.meeshogcp.in/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "farmiso-fb" - team: "sre" - service: "vmagent-farmiso-a-prd-fb" - env: "prd" - priority: "p0" - type: "vmagent-fb" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-farmiso-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-farmiso-a-prd-fb.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -# Modify resources -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - requests: - cpu: 2 - memory: 1Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - cloud.google.com/compute-class: vmagent-mds - -tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: vmagent-mds - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/gke-farmiso-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml deleted file mode 100644 index c3a5d54..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert-dr - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-farmiso-prd-dr - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/farmiso/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-farmiso-prd-dr.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-farmiso-prd-dr" - env: "prd" - priority: "p0" - type: "vmalert-dr" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - dedicated: "vmagent-dr" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-farmiso-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml deleted file mode 100644 index 7dd75ec..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-farmiso-a-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-farmiso-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-farmiso-a-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-farmiso-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/farmiso/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-farmiso-a-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "farmiso" - team: "sre" - service: "vmalert-farmiso-a-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-farmiso-prd-ase1a/victoriametrics-agent/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index 17c6695..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,482 +0,0 @@ -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "farmiso-a-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -# -- Replica count -replicaCount: 2 - -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - -# -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables -deployment: - spec: - minReadySeconds: 1200 - progressDeadlineSeconds: 1800 - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - -image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - # -- Image tag, set to `Chart.AppVersion` by default - tag: v1.133.0 # rewrites Chart.AppVersion - # -- Variant of the image to use. - # e.g. enterprise, scratch - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - -# -- Image pull secrets -imagePullSecrets: [] - -# -- Add additional DNS entries to pods hosts file. Check [official documentation](https://kubernetes.io/docs/tasks/network/customize-hosts-file-for-pods/) -hostAliases: [] -# - ip: 192.168.1.1 -# hostNames: -# - test.example.com -# - another.example.net - -# -- Override chart name -nameOverride: "" - -# -- Override resources fullname -fullnameOverride: "vm-agent-farmiso-a-prd" - -# -- Container working directory -containerWorkingDir: "/" - -rbac: - # -- Enables Role/RoleBinding creation - create: true - - # -- Role/RoleBinding annotations - annotations: {} - - # -- Role/RoleBinding labels - extraLabels: {} - - # -- If true and `rbac.enabled`, will deploy a Role/RoleBinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - - # -- additional rules for a role - extraRules: [] - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - # -- Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-farm-sre-vmagnt-prd-mds@meesho-farmiso-prd-0622.iam.gserviceaccount.com - } - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # -- mount API token to pod directly - automountToken: true - -# -- See `kubectl explain poddisruptionbudget.spec` for more or check [official documentation](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# -- Generates `remoteWrite.*` flags and config maps with value content for values, that are of type list of map. -# Each item should contain `url` param to pass validation. -remoteWrite: - # - url: http://vm-insert-farmiso-prd.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/api/v1/write - # disableOnDiskQueue: false - # dropSamplesOnOverload: false - - url: http://vm-insert-farmiso-a-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write - disableOnDiskQueue: false - dropSamplesOnOverload: false - -# -- VMAgent extra command line arguments -extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8429 - promscrape.config.strictParse: false # default is true - promscrape.maxScrapeSize: 1000000000 # default is 167772160 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - # promscrape.suppressScrapeErrors: true # default is false - - # Uncomment and specify the port if you want to support any of the protocols: - # https://docs.victoriametrics.com/victoriametrics/vmagent/#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for more details. -env: - - name: GOGC - value: "200" - - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth-secret - # key: password - -# -- Specify alternative source for env variables -envFrom: - [] - #- configMapRef: - # name: special-config - -# -- Extra labels for Deployment and Statefulset -extraLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "vm-agent-farmiso-a-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra labels for Pods only -podLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "vm-agent-farmiso-a-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra selector labels common for pod and service -selectorLabels: {} - -# -- Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# -- Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# -- Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -# -- Extra containers to run in a pod with vmagent -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -# -- Init containers for vmagent -initContainers: - [] - # - name: example - # image: example-image - -# -- Security context to be added to pod -podSecurityContext: - enabled: true - # fsGroup: 2000 - -# -- Security context to be added to pod's containers -securityContext: - enabled: true - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - # -- Enable agent service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - extraLabels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) for details - externalIPs: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8429 - # -- Target port - targetPort: http - # nodePort: 30000 - # -- Service type - type: ClusterIP - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - -ingress: - # -- Enable deployment of ingress for agent - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-agent-farmiso-a-prd.meeshogcp.in - path: - - / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - - # -- Ingress controller class name - ingressClassName: "nginx-internal" - - # -- Ingress path type - pathType: Prefix - -# -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) -resources: - requests: - cpu: 4 - memory: 8Gi - -# -- Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) -nodeSelector: - cloud.google.com/compute-class: "vmagent-mds" - -# -- Node tolerations for server scheduling to nodes with taints. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-mds" - effect: "NoSchedule" - -# -- Pod topologySpreadConstraints -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# -- Pod affinity -affinity: {} - -# -- VMAgent [scraping configuration](https://docs.victoriametrics.com/victoriametrics/vmagent/#how-to-collect-metrics-in-prometheus-format) -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-farmiso-a-prd" - -# -- Priority class to be assigned to the pod(s) -priorityClassName: "" - -# -- Enable the host network -hostNetwork: false - -serviceMonitor: - # -- Enable deployment of Service Monitor for server component. This is Prometheus operator object - enabled: false - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Service Monitor relabelings - relabelings: [] - # -- Basic auth params for Service Monitor - basicAuth: {} - # -- Service Monitor metricRelabelings - metricRelabelings: [] - # -- Service Monitor targetPort - targetPort: http - # interval: 15s - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - -# -- Empty dir configuration for a case, when persistence is disabled -emptyDir: {} - -persistentVolume: - # -- Create/use Persistent Volume Claim for server component. Empty dir if false - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Size of the volume. Should be calculated based on the logs you send and retention policy you set. - size: 10Gi - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume additional labels - extraLabels: {} - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Horizontal Pod Autoscaling. -# Note that it is not intended to be used for vmagents which perform scraping. -# In order to scale scraping vmagents check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) -horizontalPodAutoscaling: - # -- Use HPA for vmagent - enabled: false - # -- Maximum replicas for HPA to use to to scale vmagent - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale vmagent - minReplicas: 1 - # -- Metric for HPA to use to scale vmagent - metrics: [] - -# -- VMAgent scrape configuration -config: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -probe: - # -- Readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - # -- Liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - # -- Startup probe - startup: {} - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -allowedMetricsEndpoints: - - /metrics - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-farmiso-prd-ase1a/victoriametrics-insert/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/victoriametrics-insert/custom-values.yaml deleted file mode 100644 index 10dee5f..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/victoriametrics-insert/custom-values.yaml +++ /dev/null @@ -1,411 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: false - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- IDs of vmstorage nodes to exclude from writing - excludeStorageIDs: [] - # -- Override default `app` label name - name: vminsert - # -- VMInsert strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - # override Chart.AppVersion - tag: v1.107.0-cluster - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMInsert http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-farmiso-a-prd - # -- Extra command line arguments for vminsert component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8480 - maxLabelsPerTimeseries: 55 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-farmiso-a-prd-0.vm-storage-farmiso-a-prd.victoriametrics.svc:8400" - - "vm-storage-farmiso-a-prd-1.vm-storage-farmiso-a-prd.victoriametrics.svc:8400" - - "vm-storage-farmiso-a-prd-2.vm-storage-farmiso-a-prd.victoriametrics.svc:8400" - - "vm-storage-farmiso-a-prd-3.vm-storage-farmiso-a-prd.victoriametrics.svc:8400" - - "vm-storage-farmiso-a-prd-4.vm-storage-farmiso-a-prd.victoriametrics.svc:8400" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "vm-insert-farmiso-a-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Pod's additional labels - podLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "vm-insert-farmiso-a-prd" - env: "prd" - priority: "p0" - type: "vminsert" - terminationGracePeriodSeconds: 30 - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - - # -- Readiness & Liveness probes - probe: - # -- VMInsert readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert startup probe - startup: {} - - # -- Relabel configuration - relabel: - enabled: false - config: [] - # -- Use existing configmap if specified - # otherwise .config values will be used. Relabel config **should** reside under `relabel.yml` key - configMap: "" - - # Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 2 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vminsert - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vminsert - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - cloud.google.com/compute-class: "vminsert-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 2 - memory: 4Gi - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - service: - # -- Create VMInsert service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here]( https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) for details - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-insert-farmiso-a-prd.meeshogcp.in - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - - # -- Ingress controller class name - ingressClassName: nginx-internal - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for insert component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/insert" .Values.vminsert }}' - - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vminsert component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-farmiso-prd-ase1a/victoriametrics-select/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/victoriametrics-select/custom-values.yaml deleted file mode 100644 index c2333d0..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/victoriametrics-select/custom-values.yaml +++ /dev/null @@ -1,448 +0,0 @@ - -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Enable split services for vmselect (extra Service objects will be created by templates/service-split.yaml) - splitService: true - externalService: - enabled: true - name: vmselect-farmiso-a-prd-proxy - # -- Override default `app` label name - clusternativeService: - enabled: false - name: "" - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # override Chart.AppVersion - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMSelect http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-farmiso-a-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - # -- Extra command line arguments for vmselect component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8481 - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 180s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 96 - storageNode: - - "vm-storage-farmiso-a-prd-0.vm-storage-farmiso-a-prd.victoriametrics.svc:8401" - - "vm-storage-farmiso-a-prd-1.vm-storage-farmiso-a-prd.victoriametrics.svc:8401" - - "vm-storage-farmiso-a-prd-2.vm-storage-farmiso-a-prd.victoriametrics.svc:8401" - - "vm-storage-farmiso-a-prd-3.vm-storage-farmiso-a-prd.victoriametrics.svc:8401" - - "vm-storage-farmiso-a-prd-4.vm-storage-farmiso-a-prd.victoriametrics.svc:8401" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "vm-select-farmiso-a-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # -- Pod's additional labels - podLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "vm-select-farmiso-a-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Readiness & Liveness probes - probe: - # -- VMSelect readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 15 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 3 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vmselect - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vmselect - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 3 - memory: 8Gi - - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Create VMSelect service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: false - - # -- Ingress annotations - annotations: {} - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-select-farmiso-a.prd.meesho.int - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # -- Ingress controller class name - ingressClassName: contour-internal-0 - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for select component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/select" .Values.vmselect }}' - - # -- vmselect mode: deployment, daemonSet - mode: deployment - - # -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables - deployment: - spec: - # -- VMSelect strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - - # -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables - statefulSet: - enabled: false - spec: - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume extra labels - extraLabels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmselect component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: false - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - diff --git a/helm-overrides/gke-farmiso-prd-ase1a/victoriametrics-storage/custom-values.yaml b/helm-overrides/gke-farmiso-prd-ase1a/victoriametrics-storage/custom-values.yaml deleted file mode 100644 index 3364da7..0000000 --- a/helm-overrides/gke-farmiso-prd-ase1a/victoriametrics-storage/custom-values.yaml +++ /dev/null @@ -1,363 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - enabled: false - -vminsert: - enabled: false - -vmauth: - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- Override default `app` label name - name: vmstorage - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Variant of the image to use. e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMStorage http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-farmiso-a-prd - - # -- Additional environment variables (ex.: secret tokens, flags). Check https://docs.victoriametrics.com/victoriametrics/#environment-variables for details - env: [] - # -- Specify alternative source for env variables - envFrom: [] - # -- Data retention period. Possible units character: h(ours), d(ays), w(eeks), y(ears), if no unit character specified - month. The minimum retention period is 24h. - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - httpListenAddr: :8482 - loggerTimezone: "Asia/Kolkata" - dedup.minScrapeInterval: 60s - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmstorage-n4d" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - cloud.google.com/compute-class: "vmstorage-n4d" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - # -- Use an alternate scheduler, e.g. "stork". Check https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ for details - # schedulerName: - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Override Persistent Volume Claim name - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are https://kubernetes.io/docs/concepts/storage/persistent-volumes/ - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume extra labels - extraLabels: {} - # -- Storage class name. Will be empty if not set - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume - size: 250Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "vm-storage-farmiso-a-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - # -- Pod's additional labels - podLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "vm-storage-farmiso-a-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Details are https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 6 - memory: 51Gi - - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: - enabled: false - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - podSecurityContext: - enabled: false - service: - enabled: true - # -- Service annotations - annotations: {} - # -- Service ClusterIP - clusterIP: None - # -- Service type - type: ClusterIP - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmselect - vmselectPort: 8401 - # -- Extra service ports - extraPorts: [] - # -- Health check node port for a service. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilyPolicy: "" - # -- List of service IP families. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilies: [] - # -- Traffic Distribution. Check https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution - trafficDistribution: "" - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - minReadySeconds: 5 - # -- Readiness probes - probe: - # -- VMStorage readiness probe - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMStorage startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmstorage component - enabled: false - # -- Maximum replicas for HPA to use to to scale the vmstorage component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmstorage component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmstorage component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: - scaleDown: - selectPolicy: Disabled - - vmbackupmanager: - # -- Enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" \ No newline at end of file diff --git a/helm-overrides/gke-supply-prd-ase1a/README.md b/helm-overrides/gke-supply-prd-ase1a/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/gke-supply-prd-ase1a/alloy/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/alloy/custom-values.yaml deleted file mode 100644 index 938f76d..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/alloy/custom-values.yaml +++ /dev/null @@ -1,53 +0,0 @@ -fullnameOverride: "alloy-supply-a-prd" - -alloy: - configMap: - configFile: supply.alloy - - clustering: - enabled: true - - extraPorts: - - name: "otlp-grpc" - port: 4317 - targetPort: 4317 - protocol: "TCP" - - name: "otlp-http" - port: 4318 - targetPort: 4318 - protocol: "TCP" - - resources: - requests: - cpu: 7 - memory: 55Gi - limits: - cpu: 7.5 - memory: 60Gi - -configReloader: - enabled: true - -serviceAccount: - annotations: { - iam.gke.io/gcp-service-account: sa-supl-sre-grafna-obs-stk-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - } - -controller: - # Must be one of 'daemonset', 'deployment', or 'statefulset'. - type: 'deployment' - - nodeSelector: - cloud.google.com/compute-class: "alloy" - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "alloy" - effect: "NoSchedule" - - autoscaling: - enabled: true - minReplicas: 1 - maxReplicas: 50 - targetCPUUtilizationPercentage: 80 - targetMemoryUtilizationPercentage: 60 \ No newline at end of file diff --git a/helm-overrides/gke-supply-prd-ase1a/aurva-dataplane/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/aurva-dataplane/custom-values.yaml deleted file mode 100644 index 3ec53e3..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/aurva-dataplane/custom-values.yaml +++ /dev/null @@ -1,694 +0,0 @@ -## Aurva Data Plane -## Ref: https://github.com/aurva-io/aurva-charts.git - -postgresql: - enabled: true - fullnameOverride: "aurva-dataplane-database" - volumePermissions: - ## @param volumePermissions.enabled Enable init container that changes the owner and group of the persistent volume - ## - enabled: true - global: - storageClass: hyperdisk-balanced - postgresql: - auth: - postgresPassword: "aurva" - database: "controller" - # Add toleration to make sure where this postgres db pod should reside (Applicable for production workloads): For more detail ref: https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/ - primary: - extendedConfiguration: | - max_connections = 300 - - #PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: supply-devops - - # -- Select nodes to deploy which matches the following labels - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: supply-devops - -# -- Provide a name in place of `aurva` -# namespaceOverride: aurva-dataplane - -########################################################## -# Global Configs -########################################################## -global: - aurva_controller: - enabled: true - aurva_fastdet: - enabled: false - aurva_pii_analyzer: - enabled: true - aurva_ocr: - enabled: true - aurva_collector: - enabled: true - - deploymentAnnotations: {} - - priorityClassName: "" - -########################################################## -# Aurva Controller -########################################################## -aurva_controller: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "supply" - team: "supply-devops" - service: "aurva-supply-a-prd" - env: "prd" - priority: "p0" - type: "aurva_controller" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 10 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - image: - # Image of the app container - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-controller/aurva-controller - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - - type: secret - name: aurva-controller-secrets - # - type: configmap - # name: proxy-datasource-config - - # -- Resources to be defined for pod - resources: - limits: - memory: 2.5Gi - cpu: 2 - requests: - memory: 2Gi - cpu: 1 - - aurvaFastdet: - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-fastdet/aurva-fastdet - tag: "v2.30.13" - pullPolicy: IfNotPresent - envFrom: [] - env: [] - resources: - limits: - cpu: 0.5 - memory: 512Mi - requests: - cpu: 0.5 - memory: 512Mi - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: supply-devops - - # -- Taint tolerations for nodes - ##PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: supply-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-controller-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - config: - #variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - COMMAND_URL: "command.aurva-prd.meeshogcp.in:80" - DEPLOYMENT_TYPE: "kubernetes" - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - FLUSHER_WORKER_POOL_SIZE: "10000" - FLUSHER_BATCH_SIZE: "30000" - UNIQUENESS_IDENTIFIER: "k8s-supply-prd-ase1" #Recommendation: should be equal to cluster name - PROVIDER_ACCOUNT_ID: "meesho-supply-prd-0622" # GCP PROJECT ID (not Number) - REGION: "asia-southeast1" #eg: asia-south1 - ENVIRONMENT: "prod" - OCR_ENABLED: "true" - MONITORING_ENABLED: "false" - HYBRID_ONLY_MODE: "false" - FORCE_TLS: "false" - FASTDET_FLAG : "false" - AADHAAR_ENHANCER: "0" - WORKSPACE_EVENT_TRACKING_ENABLED: "false" - ACCESS_IQ_ENABLED: "true" - GRPC_ENFORCE_ALPN_ENABLED: "false" - ENABLE_FASTDET: "true" - FASTDET_MAX_BATCH_SIZE: "100" - SCAN_UUID_ENABLED: "true" - HEARTBEAT_INTERVAL: "5m" - #pii data - PII_BUCKET_NAME: "gcs-infra-devop-aurva-supply-a-prd" - PII_LOG_BUCKET_REGION: "asia-southeast1" - PII_LOG_CRON: "*/5 * * * *" - ENABLE_PII_LOG: "true" - PII_EVIDENCE_UPLOAD_MAX_BLOCKING_TASKS: "50000" - PII_EVIDENCE_UPLOAD_MAX_CONCURRENT_TASKS: "500" - PII_EVIDENCE_MAX_CACHE_WEIGHT: "100" - QUOTA_CLEANUP_CRON: "0 0 * * *" - ENABLE_PII_QUOTA: "true" - MAX_PII_EVIDENCES_PER_KEY_PER_WINDOW: "3" - PII_QUOTA_SYNC_CRON: "*/2 * * * *" - QUOTA_WINDOW_HOURS: "24" - #constants - SKIP_NAMESPACES: "contour-internal-1-supply-a-prd,contour-internal-0-supply-a-prd-intra,contour-internal-0-supply-a-prd,contour-external-supply-a-prd" - CLOUD_PROVIDER: "gcp" - LOG_ENV: "production" - RDS_SCANNER_AVAILABILITY : "false" - REDSHIFT_SCANNER_AVAILABILITY : "false" - S3_SCANNER_AVAILABILITY : "false" - DYNAMO_SCANNER_AVAILABILITY: "false" - DOCDB_SCANNER_AVAILABILITY: "false" - OPENSEARCH_SCANNER_AVAILABILITY: "false" - CLOUDSQL_SCANNER_AVAILABILITY: "true" - BIGQUERY_SCANNER_AVAILABILITY: "true" - AWS_SNAPSHOT_SCANNER_AVAILABILITY: "false" - CLOUDSTORAGE_SCANNER_AVAILABILITY: "true" - KEYSPACES_SCANNER_AVAILABILITY: "false" - ALLOYDB_SCANNER_AVAILABILITY: "true" - BIGTABLE_SCANNER_AVAILABILITY: "true" - GCP_BACKUP_AVAILABILITY: "true" - EGRESS_MODE_ONLY: "false" - SCAN_UUID_ENABLED : "true" - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-controller-sa - # -- Annotations applied to created service account - annotations: - iam.gke.io/gcp-service-account: sa-supply-prd-aurva-controller@meesho-supply-prd-0622.iam.gserviceaccount.com -# eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 15 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 - - -########################################################## -# Aurva OCR -########################################################## -aurva_ocr: - # -- Additional labels for aurva-controller - additionalLabels: - bu: "supply" - team: "supply-devops" - service: "aurva-supply-a-prd" - env: "prd" - priority: "p0" - type: "aurva_ocr" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 1 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-ocr/aurva-ocr - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-ocr: - type: secret - name: aurva-ocr-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 1 - requests: - memory: 2Gi - cpu: 1 - - # -- Select nodes to deploy which matches the following labels - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: supply-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: supply-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-ocr-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - OCR_TIME_LIMIT: "1" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "gke-supply-prd-ase1a" - DEPLOYMENT_TYPE: "kubernetes" - MAX_HOURS_TO_KEEP_REGARDLESS_OF_ACK_STATUS : "2880" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-ocr-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # iam.gke.io/gcp-service-account: service-account@gcp.iam.gserviceaccount.com - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva Collector -########################################################## -aurva_collector: - - # -- Additional labels for aurva-analyzer - additionalLabels: - bu: "supply" - team: "supply-devops" - service: "aurva-supply-a-prd" - env: "prd" - priority: "p0" - type: "aurva_collector" - - - # -- Annotations on aurva-analyzer - annotations: {} - # "key": "value" - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-collector/aurva-collector - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-controller: - type: secret - name: aurva-collector-secrets - - resources: - limits: - cpu: 200m - memory: 800Mi - requests: - cpu: 200m - memory: 512Mi - - podSecurityContext: {} - - securityContext: - privileged: true - capabilities: - add: - # For kernel v5.8 and above we don't need CAP_SYS_ADMIN or CAP_SYS_RESOURCE - # we just need CAP_BPF and CAP_PERFMON. This has been tested on our EKS node - # which is on kernel v5.10.x - # When SSL Tracing is required we need CAP_SYS_ADMIN and CAP_SYS_PTRACE - # on top of the previous capabilities - # So finally these are the 4 possible combinations for capabilies - # 1. Newer Kernels without SSL - # - BPF - # - PERFMON - # 2. Newer Kernels with SSL - - SYS_ADMIN - - SYS_PTRACE - # 3. Older Kernels without SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # 4. Older Kernels with SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # - SYS_PTRACE - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - volumes: - - name: debugfs - mountPath: /sys/kernel/debug - hostPath: /sys/kernel/debug - - name: vmlinux - mountPath: /sys/kernel/btf/vmlinux - hostPath: /sys/kernel/btf/vmlinux - - name: procfs - mountPath: /host/proc - hostPath: /proc - - name: bpffs - mountPath: /sys/fs/bpf - hostPath: /sys/fs/bpf - - # -- Taint tolerations for nodes - tolerations: - # - effect: NoSchedule - # key: cloud.google.com/compute-class - # operator: Equal - # value: megatetra - - operator: Exists - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vm-stack-ht - - vmagent-c4 - - vmagent - - vmagent-mds - - vmagent-n4 - - vminsert - - vminsert-mds - - vmselect - - vmselect-mds - - vmstorage-c4d - - vmstorage - - vmstorage-mds - - vmstorage-n4 - - vmstorage-n4d - - vmstorage-sale-24aug - - vmstorage-tmp - - contour-external - - contour-internal-0 - - contour-internal-1 - - contour-internal-1-new - - contour-intra-0 - - contour-intra-1 - - contour-shared - - alloy - - preprod-spot-16 - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-collector-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - # variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "gke-supply-prd-ase1a" - DEPLOYMENT_TYPE: "kubernetes" - TRACE_INTERNAL_SVC: "true" - TRACE_INTERNAL_SVC_HTTP: "true" - INTERNAL_SVC_SAMPLE_INTERVAL: "10m" - LOGS_TTL: "1h" - TRACE_HTTP2: "true" - TRACE_SSL: "false" - TRACE_PSQL: "false" - TRACE_SQLSERVER: "false" - TRACE_MYSQL: "false" - TRACE_EGRESS: "true" - TRACE_GO_TLS: "false" - TRACE_ML_SERVICES: "false" - # constants - LOG_ENV: production - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - MONITORING_ENABLED: "false" - ENABLE_INGRESS_INFORMER: "false" - ENABLE_SERVICE_INFORMER: "false" - ENABLE_ISTIO_INFORMER: "false" - EXCLUDED_PII_REGEX_TYPES: "ip_address,us_bank_number,us_driver_license,us_itin,us_passport,us_routing,us_mbi,ssn" - AGGREGATOR_MAX_CONNECTIONS: "1000" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-collector-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva PII Analyzer -########################################################## -aurva_pii_analyzer: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "supply" - team: "supply-devops" - service: "aurva-supply-a-prd" - env: "prd" - priority: "p0" - type: "aurva_pii_analyzer" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 16 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - ##PLACEHOLDER## - nodeSelector: - cloud.google.com/compute-class: supply-devops - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-piianalyzer/aurva-piianalyzer - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-pii-analyzer: - type: secret - name: aurva-pii-analyzer-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 5Gi - cpu: 4 - requests: - memory: 4Gi - cpu: 2 - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: supply-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: supply-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-pii-analyzer-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - SCHEDULER_TIME: "1" - SUPPORTED_REGION: "US" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "gke-supply-prd-ase1a" - DEPLOYMENT_TYPE: "kubernetes" - MAX_HOURS_TO_KEEP_REGARDLESS_OF_ACK_STATUS: "2880" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-pii-analyzer-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - - autoscaling: - enabled: true - minReplicas: 1 - maxReplicas: 2 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 diff --git a/helm-overrides/gke-supply-prd-ase1a/cert-manager/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/cert-manager/custom-values.yaml deleted file mode 100644 index e3b0c8a..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/cert-manager/custom-values.yaml +++ /dev/null @@ -1,129 +0,0 @@ -global: - logLevel: 2 - rbac: - create: true - priorityClassName: "high-priority" -installCRDs: false - -crds: - enabled: true - keep: true - -# Cert-manager Controller -replicaCount: 1 -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-controller - tag: v1.20.1 - pullPolicy: IfNotPresent - -nodeSelector: - cloud.google.com/compute-class: supply-devops - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "supply-devops" - effect: "NoSchedule" - -resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 500m - memory: 512Mi - -# Webhook Configuration -webhook: - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-webhook - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: supply-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "supply-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# CA Injector Configuration -cainjector: - enabled: true - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-cainjector - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: supply-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "supply-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# ACME Solver Configuration -acmesolver: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-acmesolver - tag: v1.20.1 - pullPolicy: IfNotPresent - -# Startup API Check -startupapicheck: - enabled: true - timeout: 1m - backoffLimit: 4 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-startupapicheck - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: supply-devops - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "supply-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - -# Prometheus Monitoring -prometheus: - enabled: true - servicemonitor: - enabled: false - interval: 60s - scrapeTimeout: 30s - labels: - prometheus: cert-manager diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/compactduo-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/compactduo-cc.yaml deleted file mode 100644 index e51877d..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/compactduo-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: compactduo -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: compactduo - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/compactocta-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/compactocta-cc.yaml deleted file mode 100644 index 6deb50b..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/compactocta-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: compactocta -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: compactocta - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-highmem-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/compacttetra-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/compacttetra-cc.yaml deleted file mode 100644 index 1542c4b..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/compacttetra-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: compacttetra -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: compacttetra - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-external-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-external-cc.yaml deleted file mode 100644 index 99372fe..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-external-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-external -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-external - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-internal-0-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-internal-0-cc.yaml deleted file mode 100644 index f357300..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-internal-0-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-0 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-internal-0 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-internal-1-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-internal-1-cc.yaml deleted file mode 100644 index 12e914b..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-internal-1-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-internal-1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-internal-1-new-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-internal-1-new-cc.yaml deleted file mode 100644 index ece1715..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-internal-1-new-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-1-new -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-internal-1-new - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-intra-0-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-intra-0-cc.yaml deleted file mode 100644 index 75e4e98..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-intra-0-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-0 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-intra-0 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-intra-1-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-intra-1-cc.yaml deleted file mode 100644 index b7bfcec..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-intra-1-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-1 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-intra-1 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-shared-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-shared-cc.yaml deleted file mode 100644 index 0437158..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/contour-shared-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-shared -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: contour-shared - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/deepgram-api-pool-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/deepgram-api-pool-cc.yaml deleted file mode 100644 index eae2ce3..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/deepgram-api-pool-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: deepgram-api-pool -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: deepgram-api-pool - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n1-standard-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/deepgram-proxy-pool-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/deepgram-proxy-pool-cc.yaml deleted file mode 100644 index 0a81e5d..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/deepgram-proxy-pool-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: deepgram-proxy-pool -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: deepgram-proxy-pool - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n1-standard-2 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/dg-eg-pool-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/dg-eg-pool-cc.yaml deleted file mode 100644 index 799f518..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/dg-eg-pool-cc.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: dg-eg-pool -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: dg-eg-pool - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - machineType: g2-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/efficient-ai-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/efficient-ai-cc.yaml deleted file mode 100644 index 1c6110d..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/efficient-ai-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: efficient-ai -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: efficient-ai - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/exp-cx-gen-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/exp-cx-gen-cc.yaml deleted file mode 100644 index 288edab..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/exp-cx-gen-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: exp-cx-gen -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: exp-cx-gen - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n2-custom-16-32768 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/g2-standard-4-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/g2-standard-4-cc.yaml deleted file mode 100644 index 2b7a547..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/g2-standard-4-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-4 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: g2-standard-4 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: g2-standard-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/megaduo-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/megaduo-cc.yaml deleted file mode 100644 index 49b6625..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/megaduo-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaduo -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megaduo - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-highcpu-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/megaduolite-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/megaduolite-cc.yaml deleted file mode 100644 index 4280cc8..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/megaduolite-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaduolite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megaduolite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-custom-16-32768 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/megaoctalite-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/megaoctalite-cc.yaml deleted file mode 100644 index 9b9a405..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/megaoctalite-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaoctalite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megaoctalite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-highmem-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n2-highmem-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/megatetra-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/megatetra-cc.yaml deleted file mode 100644 index 3214a58..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/megatetra-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megatetra -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megatetra - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-standard-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/megatetralite-azul-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/megatetralite-azul-cc.yaml deleted file mode 100644 index 3c8f163..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/megatetralite-azul-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megatetralite-azul -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megatetralite-azul - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/megatetralite-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/megatetralite-cc.yaml deleted file mode 100644 index ab21da9..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/megatetralite-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megatetralite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megatetralite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n4-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/megauno-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/megauno-cc.yaml deleted file mode 100644 index 79ddd17..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/megauno-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megauno -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megauno - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/megaunolite-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/megaunolite-cc.yaml deleted file mode 100644 index e5cff85..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/megaunolite-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: megaunolite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: megaunolite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highcpu-48 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/np-trino-n2-hmem-32-a-co-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/np-trino-n2-hmem-32-a-co-cc.yaml deleted file mode 100644 index 6ded031..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/np-trino-n2-hmem-32-a-co-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: np-trino-n2-hmem-32-a-co -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: np-trino-n2-hmem-32-a-co - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highmem-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/np-trino-n2-hmem-32-a-sp-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/np-trino-n2-hmem-32-a-sp-cc.yaml deleted file mode 100644 index a5ddb5c..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/np-trino-n2-hmem-32-a-sp-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: np-trino-n2-hmem-32-a-sp -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: np-trino-n2-hmem-32-a-sp - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highmem-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumoduo-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/sumoduo-cc.yaml deleted file mode 100644 index f88e2d3..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumoduo-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumoduo -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumoduo - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4-highcpu-48 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: c3-highcpu-44 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumoduo-cpu-mngr-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/sumoduo-cpu-mngr-cc.yaml deleted file mode 100644 index d994c2e..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumoduo-cpu-mngr-cc.yaml +++ /dev/null @@ -1,66 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumoduo-cpu-mngr -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumoduo-cpu-mngr - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4-highcpu-48 - location: - zones: ['asia-southeast1-a'] - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: c3-highcpu-44 - location: - zones: ['asia-southeast1-a'] - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: c4-highcpu-48 - location: - zones: ['asia-southeast1-c'] - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: c3-highcpu-44 - location: - zones: ['asia-southeast1-c'] - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: c4-highcpu-48 - location: - zones: ['asia-southeast1-b'] - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: c3-highcpu-44 - location: - zones: ['asia-southeast1-b'] - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumoduolite-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/sumoduolite-cc.yaml deleted file mode 100644 index 168777d..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumoduolite-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumoduolite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumoduolite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n2-custom-32-65536 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumoduolite-op-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/sumoduolite-op-cc.yaml deleted file mode 100644 index 52cdebe..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumoduolite-op-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumoduolite-op -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumoduolite-op - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n2-custom-32-65536 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumoocta-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/sumoocta-cc.yaml deleted file mode 100644 index 5801ea6..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumoocta-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumoocta -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumoocta - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-highmem-44 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumotetra-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/sumotetra-cc.yaml deleted file mode 100644 index 86c0463..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumotetra-cc.yaml +++ /dev/null @@ -1,42 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumotetra -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumotetra - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-standard-32 - location: - zones: ['asia-southeast1-a'] - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n4-standard-32 - location: - zones: ['asia-southeast1-c'] - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n4-standard-32 - location: - zones: ['asia-southeast1-b'] - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumotetra-taxonomy-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/sumotetra-taxonomy-cc.yaml deleted file mode 100644 index eddea52..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumotetra-taxonomy-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumotetra-taxonomy -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumotetra-taxonomy - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n2-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumotetra-trnst-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/sumotetra-trnst-cc.yaml deleted file mode 100644 index e256b3d..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumotetra-trnst-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumotetra-trnst -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumotetra-trnst - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumotetralite-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/sumotetralite-cc.yaml deleted file mode 100644 index 32e1f65..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumotetralite-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumotetralite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumotetralite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n2-standard-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumouno-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/sumouno-cc.yaml deleted file mode 100644 index b43c2b9..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumouno-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumouno -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumouno - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highcpu-48 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumounolite-azul-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/sumounolite-azul-cc.yaml deleted file mode 100644 index 02ed16e..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumounolite-azul-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumounolite-azul -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumounolite-azul - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n2-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumounolite-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/sumounolite-cc.yaml deleted file mode 100644 index ab471d8..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/sumounolite-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: sumounolite -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: sumounolite - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n2-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/supply-devops-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/supply-devops-cc.yaml deleted file mode 100644 index c913fc3..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/supply-devops-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: supply-devops -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: supply-devops - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: e2-standard-4 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/supply-kyverno-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/supply-kyverno-cc.yaml deleted file mode 100644 index 89ca776..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/supply-kyverno-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: supply-kyverno -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: supply-kyverno - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-standard-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/vm-stack-ht-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/vm-stack-ht-cc.yaml deleted file mode 100644 index ec1a704..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/vm-stack-ht-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vm-stack-ht -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vm-stack-ht - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highmem-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmagent-c4-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/vmagent-c4-cc.yaml deleted file mode 100644 index 373fce3..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmagent-c4-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmagent-c4 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmagent-c4 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4-highcpu-144 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmagent-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/vmagent-cc.yaml deleted file mode 100644 index 69d399c..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmagent-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmagent -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmagent - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highcpu-32 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmagent-mds-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/vmagent-mds-cc.yaml deleted file mode 100644 index deabcd1..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmagent-mds-cc.yaml +++ /dev/null @@ -1,42 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmagent-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmagent-mds - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4-highcpu-288 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n4-highcpu-48 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n4-highcpu-64 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n4-highcpu-80 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmagent-n4-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/vmagent-n4-cc.yaml deleted file mode 100644 index bde260f..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmagent-n4-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmagent-n4 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmagent-n4 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-highcpu-80 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/vminsert-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/vminsert-cc.yaml deleted file mode 100644 index bf09261..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/vminsert-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vminsert -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vminsert - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/vminsert-mds-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/vminsert-mds-cc.yaml deleted file mode 100644 index 9a902cb..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/vminsert-mds-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vminsert-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vminsert-mds - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-highcpu-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n4-highcpu-8 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmselect-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/vmselect-cc.yaml deleted file mode 100644 index 3737797..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmselect-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmselect -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmselect - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c3-highcpu-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmselect-mds-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/vmselect-mds-cc.yaml deleted file mode 100644 index 6208ed5..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmselect-mds-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmselect-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmselect-mds - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4-highcpu-24 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: c3-highcpu-22 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-c4d-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-c4d-cc.yaml deleted file mode 100644 index 05dc36b..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-c4d-cc.yaml +++ /dev/null @@ -1,25 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage-c4d -spec: - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - nodeLabels: - dedicated: vmstorage-c4d - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - priorities: - - machineType: c4d-highmem-64 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - priorityDefaults: - location: - zones: - - asia-southeast1-a - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-cc.yaml deleted file mode 100644 index e946b99..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmstorage - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highmem-48 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-mds-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-mds-cc.yaml deleted file mode 100644 index f04729a..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-mds-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage-mds -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmstorage-mds - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highmem-128 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-n4-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-n4-cc.yaml deleted file mode 100644 index de4d4bd..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-n4-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage-n4 -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmstorage-n4 - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4-highmem-64 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml deleted file mode 100644 index 1bc0ed2..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-n4d-cc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage-n4d -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmstorage-n4d - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highmem-64 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - - machineType: n4d-highmem-80 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-sale-24aug-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-sale-24aug-cc.yaml deleted file mode 100644 index f6c347f..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-sale-24aug-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage-sale-24aug -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmstorage-sale-24aug - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highmem-64 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-tmp-cc.yaml b/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-tmp-cc.yaml deleted file mode 100644 index e1eab70..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/computeclass/vmstorage-tmp-cc.yaml +++ /dev/null @@ -1,24 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: vmstorage-tmp -spec: - nodePoolConfig: - serviceAccount: sa-common-np-supl-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - nodeLabels: - dedicated: vmstorage-tmp - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2-highmem-80 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/gke-supply-prd-ase1a/conntrack-adjuster/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index c390837..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,20 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - max: 2097152 - hashsize: 524288 - sleepInterval: 30 - -nodeAffinity: - enabled: true - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: In - values: - - "contour-external" - - "contour-internal-0" - - "contour-internal-1" - - "contour-intra-0" - - "contour-intra-1" diff --git a/helm-overrides/gke-supply-prd-ase1a/contour-ca-issuer/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/contour-ca-issuer/custom-values.yaml deleted file mode 100644 index a6b1269..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/contour-ca-issuer/custom-values.yaml +++ /dev/null @@ -1,17 +0,0 @@ -# ClusterIssuer override for gke-supply-prd-ase1a (prd supply cluster, ase1a zone). -# ClusterIssuer is a cluster-scoped resource — this file pins the issuer -# identity for this cluster so the name is auditable per-cluster. -# -# Must match the `issuerRef.name` in consuming Certificate CRs (see -# devops-helm-charts/2.0.0/templates/proxyless-grpc-cert.yaml). - -issuerName: contour-supply-a-prd-ca-issuer -rootCASecretName: contour-supply-ca - -externalSecret: - enabled: true - vaultPath: meesho/devops/contour/root-ca - refreshInterval: "0" - secretStoreRef: - name: vault-backend -namespace: cert-manager-supply-prd diff --git a/helm-overrides/gke-supply-prd-ase1a/contour-cert-checker/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/contour-cert-checker/custom-values.yaml deleted file mode 100644 index 0c4fa11..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=gke-supply-prd-ase1a"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - cloud.google.com/compute-class: supply-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: supply-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/gke-supply-prd-ase1a/contour-external/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/contour-external/custom-values.yaml deleted file mode 100644 index 762a926..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/contour-external/custom-values.yaml +++ /dev/null @@ -1,112 +0,0 @@ -fullnameOverride: "contour-external-supply-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled - network: - num-trusted-hops: 1 -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: supply - team: supply-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 2048Mi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared - nodeSelector: - cloud.google.com/compute-class: contour-shared - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: supply - team: supply-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-external - nodeSelector: - cloud.google.com/compute-class: contour-external - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "40" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 6' - resources: - requests: - cpu: 6 - memory: 8Gi - limits: - cpu: 6 - memory: 12Gi - service: - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-supply-a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-supply-prd-ase1a/contour-internal-0/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/contour-internal-0/custom-values.yaml deleted file mode 100644 index 3bd0f14..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,125 +0,0 @@ -fullnameOverride: "contour-internal-0-supply-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: supply - team: supply-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 4Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared - nodeSelector: - cloud.google.com/compute-class: contour-shared - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - enabled: true - podLabels: - bu: supply - team: supply-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-0 - nodeSelector: - cloud.google.com/compute-class: contour-internal-0 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "55" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 14' - resources: - requests: - cpu: 14 - memory: 8Gi - limits: - cpu: 14 - memory: 27Gi - service: - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-supply-a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-supply-prd-ase1a/contour-internal-1/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/contour-internal-1/custom-values.yaml deleted file mode 100644 index 7634fc8..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,118 +0,0 @@ -fullnameOverride: "contour-internal-1-supply-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: supply - team: supply-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 12 - memory: 22Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-intra-0 - nodeSelector: - cloud.google.com/compute-class: contour-intra-0 - service: - tcpLB: true - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: supply - team: supply-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-1 - nodeSelector: - cloud.google.com/compute-class: contour-internal-1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 14' - resources: - requests: - cpu: 14 - memory: 6Gi - limits: - cpu: 14 - memory: 27Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int1-supply-a-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-supply-prd-ase1a/contour-internal-intra-0/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index b4a7077..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,123 +0,0 @@ -fullnameOverride: "contour-internal-0-supply-intra-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: supply - team: supply-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 4Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared - nodeSelector: - cloud.google.com/compute-class: contour-shared - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - enabled: true - podLabels: - bu: supply - team: supply-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-intra-0 - nodeSelector: - cloud.google.com/compute-class: contour-intra-0 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "55" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 14' - resources: - requests: - cpu: 14 - memory: 8Gi - limits: - cpu: 14 - memory: 27Gi - service: - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-supply-prd-ase1a/contour-internal-intra-1/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index efd0a84..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -fullnameOverride: "contour-internal-1-supply-intra-prd" -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: supply - team: supply-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 4Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared - nodeSelector: - cloud.google.com/compute-class: contour-shared - service: - tcpLB: false - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: supply - team: supply-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-intra-1 - nodeSelector: - cloud.google.com/compute-class: contour-intra-1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 6' - resources: - requests: - cpu: 6 - memory: 3Gi - limits: - cpu: 6 - memory: 12Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend" diff --git a/helm-overrides/gke-supply-prd-ase1a/coredns/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/coredns/custom-values.yaml deleted file mode 100644 index 04c7d45..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/coredns/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -replicaCount: 30 -communicationType: "intra" - -labels: - bu: supply - team: supply-devops - env: prd - -clusterIP: 10.1.16.2 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "supply-devops" - effect: "NoSchedule" - -nodeSelector: - cloud.google.com/compute-class: supply-devops - kubernetes.io/os: linux diff --git a/helm-overrides/gke-supply-prd-ase1a/coroot-node-agent/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/coroot-node-agent/custom-values.yaml deleted file mode 100644 index fb3822e..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/coroot-node-agent/custom-values.yaml +++ /dev/null @@ -1,54 +0,0 @@ -fullnameOverride: "coroot-node-agent-supply-a-prd" -priorityClassName: "low-priority" - -scrape: "false" # disable scraping for now - -resources: - requests: - cpu: "300m" - memory: "1Gi" - limits: - cpu: "500m" - memory: "1.5Gi" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-supply-contour-ext - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-supply-default-prd-ase1 - - key: cloud.google.com/compute-class - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external - - alloy - - sumoduolite-op - - sumounolite \ No newline at end of file diff --git a/helm-overrides/gke-supply-prd-ase1a/deepgram-onprem-v2/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/deepgram-onprem-v2/custom-values.yaml deleted file mode 100644 index e18e465..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/deepgram-onprem-v2/custom-values.yaml +++ /dev/null @@ -1,410 +0,0 @@ -global: - pullSecretRef: dg-regcred - deepgramSecretRef: dg-self-hosted-api-key-v2 - additionalLabels: {} - outstandingRequestGracePeriod: 1800 - -apiAutoscaling: - enabled: true - targetName: deepgram-api - maxReplicas: 500 - minReplicas: 2 - pollingInterval: 30 - scaledown: - policies: - - periodseconds: 60 - type: Pods - value: 2 - selectpolicy: Min - stabilizationWindowSeconds: 900 - scaleup: - policies: - - periodseconds: 15 - type: Pods - value: 2 - - periodseconds: 15 - type: Percent - value: 10 - selectpolicy: Max - stabilizationWindowSeconds: 60 - triggers: - - metadata: - metricName: engine_active_requests - query: sum(kube_deployment_status_replicas_available{namespace="dg-self-hosted-v2-a",deployment="deepgram-engine-a"}) - serverAddress: http://vmselect-supply-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus/ - threshold: "0.5" - type: prometheus - -engineAutoscaling: - enabled: true - targetName: deepgram-engine-a - maxReplicas: 500 - minReplicas: 2 - pollingInterval: 30 - scaledown: - policies: - - periodseconds: 60 - type: Pods - value: 2 - selectpolicy: Min - stabilizationWindowSeconds: 900 - scaleup: - policies: - - periodseconds: 15 - type: Pods - value: 2 - - periodseconds: 15 - type: Percent - value: 10 - selectpolicy: Max - stabilizationWindowSeconds: 60 - triggers: - - metadata: - metricName: engine_active_requests - query: sum(engine_active_requests{kind="stream",kubernetes_namespace="dg-self-hosted-v2-a"}) - serverAddress: http://vmselect-supply-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus/ - threshold: "5" - type: prometheus - -scaling: - replicas: - api: 15 - engine: 20 - auto: - enabled: false - api: - metrics: - engineToApiRatio: 4 - custom: - behavior: - scaleDown: - policies: - - type: Pods - value: 1 - periodSeconds: 60 - - type: Percent - value: 25 - periodSeconds: 60 - engine: - minReplicas: 1 - maxReplicas: 10 - metrics: - requestCapacityRatio: 0.8 - speechToText: - batch: - requestsPerPod: 12 - streaming: - requestsPerPod: 14 - textToSpeech: - batch: - requestsPerPod: 50 - custom: [] - behavior: - scaleDown: - policies: - - type: Pods - value: 1 - periodSeconds: 60 - - type: Percent - value: 25 - periodSeconds: 60 - -api: - createContourGateway: true - ingress: - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: 'false' - nginx.ingress.kubernetes.io/ssl-redirect: 'false' - enabled: true - hosts: - - host: deepgram-v2.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / - apiServiceName: deepgram-api-a-external - servicePort: 80 - ingressClassName: contour-internal-0 - servicePort: 80 - enableWebsocket: false - namePrefix: deepgram-api-a - namespace: dg-self-hosted-v2-a - slowStart: - enabled: true - window: 60s - aggression: 0.5 - minPercent: 5 - - image: - path: quay.io/deepgram/self-hosted-api - pullPolicy: IfNotPresent - tag: release-251118 - - additionalLabels: {} - additionalAnnotations: - - updateStrategy: - rollingUpdate: - maxUnavailable: 0 - maxSurge: 1 - - resources: - requests: - memory: "4Gi" - cpu: "2000m" - limits: - memory: "8Gi" - cpu: "4000m" - - readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 1 - livenessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 3 - - affinity: {} - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: deepgram-api-pool - - securityContext: {} - - serviceAccount: - create: true - name: - - server: - baseUrl: "/v1" - host: "0.0.0.0" - port: 8080 - callbackConnTimeout: "1s" - callbackTimeout: "10s" - fetchConnTimeout: "1s" - fetchTimeout: "60s" - - resolver: - nameservers: [] - maxTTL: - - features: - entityDetection: false - entityRedaction: false - diskBufferPath: - - driverPool: - standard: - timeoutBackoff: 1.2 - retrySleep: "2s" - retryBackoff: 1.6 - maxResponseSize: "1073741824" - -engine: - namePrefix: "deepgram-engine-a" - - image: - path: quay.io/deepgram/self-hosted-engine - pullPolicy: IfNotPresent - tag: release-251118 - - additionalLabels: {} - additionalAnnotations: - - updateStrategy: - rollingUpdate: - maxUnavailable: 0 - maxSurge: 1 - - resources: - requests: - memory: "15Gi" - cpu: "6" - gpu: 1 - limits: - memory: "20Gi" - cpu: "6" - gpu: 1 - - startupProbe: - periodSeconds: 10 - failureThreshold: 60 - - readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 1 - livenessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 3 - - lifecycle: - preStop: - exec: - command: - - /bin/bash - - -c - - /bin/sleep 30 - - affinity: {} - nodeSelector: - cloud.google.com/compute-class: dg-eg-pool - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: dg-eg-pool - - effect: NoSchedule - key: nvidia.com/gpu - operator: Equal - value: present - - effect: NoSchedule - key: nvidia.com/gpu - operator: Exists - - securityContext: {} - - serviceAccount: - create: true - name: - - concurrencyLimit: - activeRequests: 12 - - server: - host: "0.0.0.0" - port: 8080 - - metricsServer: - host: "0.0.0.0" - port: 9273 - - modelManager: - volumes: - customVolumeClaim: - enabled: false - name: - modelsDirectory: "/" - aws: - efs: - enabled: false - namePrefix: dg-models-a - fileSystemId: - forceDownload: false - nova3: - enabled: false - multilingual: - enabled: true - gcp: - gpd: - enabled: true - namePrefix: dg-models-v2-a - storageClassName: "standard-rwo" - storageCapacity: "50G" - volumeHandle: "projects/meesho-supply-prd-0622/zones/asia-southeast1-a/disks/deepgram-model-storage-nova3-multilingual-v2-a" - fsType: "ext4" - models: - links: [] - - chunking: - speechToText: - batch: - minDuration: - maxDuration: - streaming: - minDuration: - maxDuration: - step: 0.2 - - halfPrecision: - state: "auto" - -licenseProxy: - enabled: false - deploySecondReplica: false - keepUpstreamServerAsBackup: true - namePrefix: "deepgram-license-proxy-a" - - image: - path: quay.io/deepgram/self-hosted-license-proxy - tag: release-251118 - pullPolicy: IfNotPresent - - additionalLabels: {} - additionalAnnotations: - - updateStrategy: - rollingUpdate: - maxSurge: 1 - - resources: - requests: - memory: "1Gi" - cpu: "1000m" - limits: - memory: "8Gi" - cpu: "2000m" - - readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 1 - livenessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 3 - - affinity: {} - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: deepgram-proxy-pool - - securityContext: {} - - serviceAccount: - create: true - name: - - server: - host: "0.0.0.0" - port: 8443 - baseUrl: "/" - statusPort: 8080 - -gpu-operator: - enabled: false - driver: - enabled: true - version: "550.54.15" - toolkit: - enabled: true - version: v1.15.0-ubi8 - -cluster-autoscaler: - enabled: false - -kube-prometheus-stack: - includeDependency: false - fullnameOverride: "dg-prometheus-stack" - prometheusOperator: - enabled: false - alertmanager: - enabled: false - grafana: - enabled: false - nodeExporter: - enabled: false - kube-state-metrics: - enabled: false - -prometheus-adapter: - includeDependency: - prometheus: - url: http://dg-prometheus-stack-prometheus.{{ .Release.Namespace }}.svc - rules: - default: false diff --git a/helm-overrides/gke-supply-prd-ase1a/deepgram-onprem/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/deepgram-onprem/custom-values.yaml deleted file mode 100644 index 9069fac..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/deepgram-onprem/custom-values.yaml +++ /dev/null @@ -1,918 +0,0 @@ -global: - # -- (string) If using images from the Deepgram Quay image repositories, - # or another private registry to which your cluster doesn't have default access, - # you will need to provide a pre-configured K8s Secret - # with image repository credentials. See chart docs for more details. - pullSecretRef: dg-regcred - - # -- (string) Name of the pre-configured K8s Secret containing your Deepgram - # self-hosted API key. See chart docs for more details. - deepgramSecretRef: dg-self-hosted-api-key-v2 - - # -- Additional labels to add to all Deepgram resources - additionalLabels: {} - - # -- When an API or Engine container is signaled to shutdown via Kubernetes sending a SIGTERM - # signal, the container will stop listening on its port, and no new requests will be routed - # to that container. However, the container will continue to run until all existing - # batch or streaming requests have completed, after which it will gracefully shut down. - # - # Batch requests should be finished within 10-15 minutes, but streaming requests can proceed indefinitely. - # - # outstandingRequestGracePeriod defines the period (in sec) after which Kubernetes will forcefully - # shutdown the container, terminating any outstanding connections. 1800 / 60 sec/min = 30 mins - outstandingRequestGracePeriod: 1800 - -# -- Configuration options for horizontal scaling of Deepgram -# services. Only one of `static` and `auto` options can be enabled. -# @default -- `` - -apiAutoscaling: - enabled: true - targetName: deepgram-api - maxReplicas: 500 - minReplicas: 6 - pollingInterval: 30 - scaledown: - policies: - - periodseconds: 60 - type: Pods - value: 2 - selectpolicy: Min - stabilizationWindowSeconds: 900 - scaleup: - policies: - - periodseconds: 15 - type: Pods - value: 2 - - periodseconds: 15 - type: Percent - value: 10 - selectpolicy: Max - stabilizationWindowSeconds: 60 - triggers: - - metadata: - metricName: engine_active_requests - query: sum(kube_deployment_status_replicas_available{namespace="dg-self-hosted-a",deployment="deepgram-engine-a"}) - serverAddress: http://vmselect-supply-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus/ - threshold: "0.5" - type: prometheus - - metadata: - desiredReplicas: "10" - end: "0 7 * * *" - start: "30 23 * * *" - timezone: "Asia/Kolkata" - type: cron - - - -engineAutoscaling: - enabled: true - targetName: deepgram-engine-a - maxReplicas: 500 - minReplicas: 3 - pollingInterval: 30 - scaledown: - policies: - - periodseconds: 60 - type: Pods - value: 2 - selectpolicy: Min - stabilizationWindowSeconds: 900 - scaleup: - policies: - - periodseconds: 15 - type: Pods - value: 2 - - periodseconds: 15 - type: Percent - value: 10 - selectpolicy: Max - stabilizationWindowSeconds: 60 - triggers: - - metadata: - metricName: engine_active_requests - query: sum(engine_active_requests{kind="stream",kubernetes_namespace="dg-self-hosted-a"}) - serverAddress: http://vmselect-supply-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus/ - threshold: "5" - type: prometheus - - metadata: - desiredReplicas: "5" - end: "0 7 * * *" - start: "30 23 * * *" - timezone: "Asia/Kolkata" - type: cron - -scaling: - # -- Number of replicas to set during initial installation. - # @default -- `` - replicas: - api: 15 - engine: 20 - - # -- Enable pod autoscaling based on system load/traffic. - # @default -- `` - auto: - enabled: false - - api: - metrics: - # -- Scale the API deployment to this Engine-to-Api pod ratio - engineToApiRatio: 4 - # -- (list) If you have custom metrics you would like to scale with, you may add them here. - # See the [k8s docs](https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/) - # for how to structure a list of metrics - custom: - - # -- [Configurable scaling behavior](https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#configurable-scaling-behavior) - # @default -- "*See values.yaml file for default*" - behavior: - scaleDown: - policies: - - type: Pods - value: 1 - periodSeconds: 60 - - type: Percent - value: 25 - periodSeconds: 60 - - engine: - # -- Minimum number of Engine replicas. - minReplicas: 1 - # -- Maximum number of Engine replicas. - maxReplicas: 10 - metrics: - # -- If `engine.concurrencyLimit.activeRequests` is set, this variable will - # define the ratio of current active requests to maximum active requests at which - # the Engine pods will scale. Setting this value too close to 1.0 may lead to a situation where - # the cluster is at max capacity and rejects incoming requests. Setting the ratio too close to 0.0 - # will over-optimistically scale your cluster and increase compute costs unnecessarily. - requestCapacityRatio: 0.8 - speechToText: - batch: - # -- (int) Scale the Engine pods based on a static desired number of speech-to-text batch requests per pod - requestsPerPod: 12 - streaming: - # -- (int) Scale the Engine pods based on a static desired number of speech-to-text streaming requests per pod - requestsPerPod: 14 - textToSpeech: - batch: - # -- (int) Scale the Engine pods based on a static desired number of text-to-speech batch requests per pod - requestsPerPod: 50 - # -- If you have custom metrics you would like to scale with, you may add them here. - # See the [k8s docs](https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/) - # for how to structure a list of metrics - custom: [] - - # -- [Configurable scaling behavior](https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#configurable-scaling-behavior) - # @default -- "*See values.yaml file for default*" - behavior: - scaleDown: - policies: - - type: Pods - value: 1 - periodSeconds: 60 - - type: Percent - value: 25 - periodSeconds: 60 - -api: - # -- namePrefix is the prefix to apply to the name of all K8s objects - # associated with the Deepgram API containers. - createContourGateway: true - ingress: - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: 'false' - nginx.ingress.kubernetes.io/ssl-redirect: 'false' - enabled: true - hosts: - - host: deepgram.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / - apiServiceName: deepgram-api-a-external - servicePort: 80 - ingressClassName: contour-internal-1 - servicePort: 80 - enableWebsocket: false - namePrefix: deepgram-api-a - namespace: dg-self-hosted-a - slowStart: - enabled: true - window: 60s - aggression: 0.5 - minPercent: 5 - - - image: - # -- path configures the image path to use for creating API containers. - # You may change this from the public Quay image path if you have imported - # Deepgram images into a private container registry. - path: quay.io/deepgram/self-hosted-api - # -- pullPolicy configures how the Kubelet attempts to pull the Deepgram API image - pullPolicy: IfNotPresent - # -- tag defines which Deepgram release to use for API containers - tag: release-251118 - - # -- Additional labels to add to API resources - additionalLabels: {} - - # -- (object) Additional annotations to add to the API deployment - additionalAnnotations: - - updateStrategy: - rollingUpdate: - # -- The maximum number of API pods, relative to the number of replicas, - # that can go offline during a rolling update. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-unavailable) - # for more details. - maxUnavailable: 0 - # -- The maximum number of extra API pods that can be created during a rollingUpdate, - # relative to the number of replicas. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-surge) - # for more details. - maxSurge: 1 - - # -- Configure resource limits per API container. See - # [Deepgram's documentation](https://developers.deepgram.com/docs/self-hosted-deployment-environments#api) - # for more details. - # @default -- `` - resources: - requests: - memory: "4Gi" - cpu: "2000m" - limits: - memory: "8Gi" - cpu: "4000m" - - # -- Readiness probe customization for API pods. - # @default -- `` - readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 1 - # -- Liveness probe customization for API pods. - # @default -- `` - livenessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 3 - - # -- [Affinity and anti-affinity](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#affinity-and-anti-affinity) - # to apply for API pods. - affinity: {} - # -- [Tolerations](https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/) - # to apply to API pods. - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: deepgram-api-pool - - - # -- [Security context](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) for API pods. - securityContext: {} - - serviceAccount: - # -- Specifies whether to create a default service account for the Deepgram API Deployment. - create: true - # -- (string) Allows providing a custom service account name for the API component. - # If left empty, the default service account name will be used. - # If specified, and `api.serviceAccount.create = true`, this defines the name of the default service account. - # If specified, and `api.serviceAccount.create = false`, this provides the name of a preconfigured service account - # you wish to attach to the API deployment. - name: - - # -- Configure how the API will listen for your requests - # @default -- `` - server: - # baseUrl is the prefix requests to the API. - baseUrl: "/v1" - # -- host is the IP address to listen on. You will want to listen - # on all interfaces to interact with other pods in the cluster. - host: "0.0.0.0" - # -- port to listen on. - port: 8080 - - # -- callbackConnTimeout configures how long to wait for a connection to a callback URL. - # See [Deepgram's callback documentation](https://developers.deepgram.com/docs/callback) - # for more details. The value should be a humantime duration. - callbackConnTimeout: "1s" - # -- callbackTimeout configures how long to wait for a response from a callback URL. - # See [Deepgram's callback documentation](https://developers.deepgram.com/docs/callback) - # for more details. The value should be a humantime duration. - callbackTimeout: "10s" - - # -- fetchConnTimeout configures how long to wait for a connection to a fetch URL. - # The value should be a humantime duration. - # A fetch URL is a URL passed in an inference request from which a payload should be - # downloaded. - fetchConnTimeout: "1s" - # -- fetchTimeout configures how long to wait for a response from a fetch URL. - # The value should be a humantime duration. - # A fetch URL is a URL passed in an inference request from which a payload should be - # downloaded. - fetchTimeout: "60s" - - # -- Specify custom DNS resolution options. - # @default -- `` - resolver: - # -- nameservers allows for specifying custom domain name server(s). - # A valid list item's format is "{IP} {PORT} {PROTOCOL (tcp or udp)}", - # e.g. `"127.0.0.1 53 udp"`. - nameservers: [] - # -- (int) maxTTL sets the DNS TTL value if specifying a custom DNS nameserver. - maxTTL: - - # -- Enable ancillary features - # @default -- `` - features: - # -- Enables entity detection on pre-recorded audio - # *if* a valid entity detection model is available. - # *WARNING*: Beta functionality. - entityDetection: false - - # -- Enables entity-based redaction on pre-recorded audio - # *if* a valid entity detection model is available. - # *WARNING*: Beta functionality. - entityRedaction: false - - # -- If API is receiving requests faster than Engine can process them, a request - # queue will form. By default, this queue is stored in memory. Under high load, - # the queue may grow too large and cause Out-Of-Memory errors. To avoid this, - # set a diskBufferPath to buffer the overflow on the request queue to disk. - # - # WARN: This is only to temporarily buffer requests during high load. - # If there is not enough Engine capacity to process the queued requests over time, - # the queue (and response time) will grow indefinitely. - diskBufferPath: - - # -- driverPool configures the backend pool of speech engines (generically referred to as - # "drivers" here). The API will load-balance among drivers in the standard - # pool; if one standard driver fails, the next one will be tried. - # @default -- `` - driverPool: - # -- standard is the main driver pool to use. - # @default -- `` - standard: - # -- timeoutBackoff is the factor to increase the timeout by - # for each additional retry (for exponential backoff). - timeoutBackoff: 1.2 - - # -- retrySleep defines the initial sleep period (in humantime duration) - # before attempting a retry. - retrySleep: "2s" - # -- retryBackoff is the factor to increase the retrySleep - # by for each additional retry (for exponential backoff). - retryBackoff: 1.6 - - # -- Maximum response to deserialize from Driver (in bytes). - # Default is 1GB, expressed in bytes. - maxResponseSize: "1073741824" - -engine: - # -- namePrefix is the prefix to apply to the name of all K8s objects - # associated with the Deepgram Engine containers. - namePrefix: "deepgram-engine-a" - - image: - # -- path configures the image path to use for creating Engine containers. - # You may change this from the public Quay image path if you have imported - # Deepgram images into a private container registry. - path: quay.io/deepgram/self-hosted-engine - # -- pullPolicy configures how the Kubelet attempts to pull the Deepgram Engine image - pullPolicy: IfNotPresent - # -- tag defines which Deepgram release to use for Engine containers - tag: release-251118 - - # -- Additional labels to add to Engine resources - additionalLabels: {} - - # -- (object) Additional annotations to add to the Engine deployment - additionalAnnotations: - - updateStrategy: - rollingUpdate: - # -- The maximum number of Engine pods, relative to the number of replicas, - # that can go offline during a rolling update. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-unavailable) - # for more details. - maxUnavailable: 0 - # -- The maximum number of extra Engine pods that can be created during a rollingUpdate, - # relative to the number of replicas. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-surge) - # for more details. - maxSurge: 1 - - # -- Configure resource limits per Engine container. See - # [Deepgram's documentation](https://developers.deepgram.com/docs/self-hosted-deployment-environments#engine) - # for more details. - # @default -- `` - resources: - requests: - memory: "15Gi" - cpu: "6" - # -- gpu maps to the nvidia.com/gpu resource parameter - gpu: 1 - limits: - memory: "20Gi" - cpu: "6" - # -- gpu maps to the nvidia.com/gpu resource parameter - gpu: 1 - - # -- The startupProbe combination of `periodSeconds` and `failureThreshold` allows - # time for the container to load all models and start listening for incoming requests. - # - # Model load time can be affected by hardware I/O speeds, as well as network speeds - # if you are using a network volume mount for the models. - # - # If you are hitting the failure threshold before models are finished loading, you may - # want to extend the startup probe. However, this will also extend the time it takes - # to detect a pod that can't establish a network connection to validate its license. - # @default -- `` - startupProbe: - # -- periodSeconds defines how often to execute the probe. - periodSeconds: 10 - # -- failureThreshold defines how many unsuccessful startup probe attempts - # are allowed before the container will be marked as Failed - failureThreshold: 60 - - # -- Readiness probe customization for Engine pods. - # @default -- `` - readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 1 - # -- Liveness probe customization for Engine pods. - # @default -- `` - livenessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 3 - - # -- [Container lifecycle hooks](https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/) - lifecycle: - preStop: - exec: - command: - - /bin/bash - - -c - - /bin/sleep 30 - - # -- [Affinity and anti-affinity](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#affinity-and-anti-affinity) - # to apply for Engine pods. - affinity: {} - # -- [Tolerations](https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/) - # to apply to Engine pods. - nodeSelector: - cloud.google.com/compute-class: dg-eg-pool - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: dg-eg-pool - - effect: NoSchedule - key: nvidia.com/gpu - operator: Equal - value: present - - effect: NoSchedule - key: nvidia.com/gpu - operator: Exists - - - # -- [Security context](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) for API pods. - securityContext: {} - - serviceAccount: - # -- Specifies whether to create a default service account for the Deepgram Engine Deployment. - create: true - # -- (string) Allows providing a custom service account name for the Engine component. - # If left empty, the default service account name will be used. - # If specified, and `engine.serviceAccount.create = true`, this defines the name of the default service account. - # If specified, and `engine.serviceAccount.create = false`, this provides the name of a preconfigured service account - # you wish to attach to the Engine deployment. - name: - - concurrencyLimit: - # -- (int) activeRequests limits the number of active requests handled by - # a single Engine container. - # If additional requests beyond the limit are sent, the API container - # forming the request will try a different Engine pod. If no Engine pods - # are able to accept the request, the API will return a 429 HTTP response - # to the client. The `nil` default means no limit will be set. - activeRequests: 12 - - # -- Configure Engine containers to listen for requests from API containers. - # @default -- `` - server: - # -- host is the IP address to listen on for inference requests. - # You will want to listen on all interfaces to interact with - # other pods in the cluster. - host: "0.0.0.0" - # -- port to listen on for inference requests - port: 8080 - - # -- metricsServer exposes an endpoint on each Engine container - # for reporting inference-specific system metrics. - # See https://developers.deepgram.com/docs/metrics-guide#deepgram-engine - # for more details. - # @default -- `` - metricsServer: - # -- host is the IP address to listen on for metrics requests. - # You will want to listen on all interfaces to interact with - # other pods in the cluster. - host: "0.0.0.0" - # -- port to listen on for metrics requests - port: 9273 - - modelManager: - volumes: - customVolumeClaim: - # -- You may manually create your own PersistentVolume and PersistentVolumeClaim to store and - # expose model files to the Deepgram Engine. Configure your storage beforehand, - # and enable here. - # Note: Make sure the PV and PVC accessMode are set to `readWriteMany` or `readOnlyMany` - enabled: false - # -- (string) Name of your pre-configured PersistentVolumeClaim - name: - # -- Name of the directory within your pre-configured PersistentVolume - # where the models are stored - modelsDirectory: "/" - - aws: - efs: - # -- Whether to use an [AWS Elastic File Sytem](https://aws.amazon.com/efs/) - # to store Deepgram models for use by Engine containers. - # This option requires your cluster to be running in - # [AWS EKS](https://aws.amazon.com/eks/). - enabled: false - # -- Name prefix for the resources associated with the model storage in AWS EFS. - namePrefix: dg-models-a - # -- (string) FileSystemId of existing AWS Elastic File System where - # Deepgram model files will be persisted. - # You can find it using the AWS CLI: - # ``` - # $ aws efs describe-file-systems --query "FileSystems[*].FileSystemId" - # ``` - fileSystemId: - # -- Whether to force a fresh download of all model links provided, - # even if models are already present in EFS. - forceDownload: false - nova3: - enabled: false - multilingual: - enabled: true - gcp: - gpd: - # -- Whether to use an [GKE Persistent Disks](https://cloud.google.com/kubernetes-engine/docs/concepts/persistent-volumes) - # to store Deepgram models for use by Engine containers. - # This option requires your cluster to be running in - # [GCP GKE](https://cloud.google.com/kubernetes-engine). - # See the GKE documentation on - # [using pre-existing persistent disks](https://cloud.google.com/kubernetes-engine/docs/how-to/persistent-volumes/preexisting-pd). - enabled: true - # -- Name prefix for the resources associated with the model storage in GCP GPD. - namePrefix: dg-models-a - # -- The storageClassName of the existing persistent disk. - storageClassName: "standard-rwo" - # -- The size of your pre-existing persistent disk. - storageCapacity: "50G" - # -- The identifier of your pre-existing persistent disk. - # The format is projects/{project_id}/zones/{zone_name}/disks/{disk_name} for Zonal persistent disks, - # or projects/{project_id}/regions/{region_name}/disks/{disk_name} for Regional persistent disks. - volumeHandle: "projects/meesho-supply-prd-0622/zones/asia-southeast1-a/disks/deepgram-model-storage-nova3-multilingual-a" - fsType: "ext4" - - models: - # -- Links to your Deepgram models, if automatically downloading - # into storage backing a persistent volume. - # **Automatic downloads are currently supported for AWS EFS volumes only.** - # Insert each model link provided to you by your Deepgram - # Account Representative. - links: [] - - # -- chunking defines the size of audio chunks to process in seconds. - # Adjusting these values will affect both inference performance and accuracy - # of results. Please contact your Deepgram Account Representative if you - # want to adjust any of these values. - # @default -- `` - chunking: - speechToText: - batch: - # -- (float) minDuration is the minimum audio duration for a STT chunk size for a batch request - minDuration: - # -- (float) minDuration is the maximum audio duration for a STT chunk size for a batch request - maxDuration: - streaming: - # -- (float) minDuration is the minimum audio duration for a STT chunk size for a streaming request - minDuration: - # -- (float) minDuration is the maximum audio duration for a STT chunk size for a streaming request - maxDuration: - # -- step defines how often to return interim results, in seconds. - # This value may be lowered to increase the frequency of interim results. - # However, this also causes a significant decrease in the number of concurrent - # streams supported by a single GPU. Please contact your Deepgram Account - # representative for more details. - step: 0.2 - - halfPrecision: - # -- Engine will automatically enable half precision operations if your GPU supports - # them. You can explicitly enable or disable this behavior with the state parameter - # which supports `"enable"`, `"disabled"`, and `"auto"`. - state: "auto" - -# -- Configuration options for the optional -# [Deepgram License Proxy](https://developers.deepgram.com/docs/license-proxy). -# @default -- `` -licenseProxy: - # -- The License Proxy is optional, but highly recommended to be deployed in production - # to enable highly available environments. - enabled: false - - # -- If the License Proxy is deployed, one replica should be sufficient to - # support many API/Engine pods. - # Highly available environments may wish to deploy a second replica to ensure - # uptime, which can be toggled with this option. - deploySecondReplica: false - - # -- Even with a License Proxy deployed, API and Engine pods can be configured to keep the - # upstream `license.deepgram.com` license server as a fallback licensing option if the - # License Proxy is unavailable. - # Disable this option if you are restricting API/Engine Pod network access for security reasons, - # and only the License Proxy should send egress traffic to the upstream license server. - keepUpstreamServerAsBackup: true - - # -- namePrefix is the prefix to apply to the name of all K8s objects - # associated with the Deepgram License Proxy containers. - namePrefix: "deepgram-license-proxy-a" - - image: - # -- path configures the image path to use for creating License Proxy containers. - # You may change this from the public Quay image path if you have imported - # Deepgram images into a private container registry. - path: quay.io/deepgram/self-hosted-license-proxy - # -- tag defines which Deepgram release to use for License Proxy containers - tag: release-251118 - # -- pullPolicy configures how the Kubelet attempts to pull the Deepgram - # License Proxy image - pullPolicy: IfNotPresent - - # -- Additional labels to add to License Proxy resources - additionalLabels: {} - - # -- (object) Additional annotations to add to the LicenseProxy deployment - additionalAnnotations: - - updateStrategy: - # -- For the LicenseProxy, we only expose maxSurge and not maxUnavailable. - # This is to avoid accidentally having all LicenseProxy nodes go offline during upgrades, - # which could impact the entire cluster's connection to the Deepgram License Server. - # @default -- `` - rollingUpdate: - # -- The maximum number of extra License Proxy pods that can be created during a rollingUpdate, - # relative to the number of replicas. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-surge) - # for more details. - maxSurge: 1 - - # -- Configure resource limits per License Proxy container. See - # [Deepgram's documentation](https://developers.deepgram.com/docs/license-proxy#system-requirements) - # for more details. - # @default -- `` - resources: - requests: - memory: "1Gi" - cpu: "1000m" - limits: - memory: "8Gi" - cpu: "2000m" - - # -- Readiness probe customization for License Proxy pods. - # @default -- `` - readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 1 - # -- Liveness probe customization for Proxy pods. - # @default -- `` - livenessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 3 - - # -- [Affinity and anti-affinity](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#affinity-and-anti-affinity) - # to apply for License Proxy pods. - affinity: {} - # -- [Tolerations](https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/) - # to apply to License Proxy pods. - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: deepgram-proxy-pool - - # -- [Security context](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) for API pods. - securityContext: {} - - serviceAccount: - # -- Specifies whether to create a default service account for the Deepgram License Proxy Deployment. - create: true - # -- (string) Allows providing a custom service account name for the LicenseProxy component. - # If left empty, the default service account name will be used. - # If specified, and `licenseProxy.serviceAccount.create = true`, this defines the name of the default service account. - # If specified, and `licenseProxy.serviceAccount.create = false`, this provides the name of a preconfigured service account - # you wish to attach to the License Proxy deployment. - name: - - # -- Configure how the license proxy will listen for licensing requests. - # @default -- `` - server: - # --host is the IP address to listen on. You will want to listen - # on all interfaces to interact with other pods in the cluster. - host: "0.0.0.0" - # -- port to listen on. - port: 8443 - - # -- baseUrl is the prefix for incoming license verification requests. - baseUrl: "/" - - # -- statusPort is the port to listen on for the status/health endpoint. - statusPort: 8080 - -# -- Passthrough values for [NVIDIA GPU Operator Helm chart](https://github.com/NVIDIA/gpu-operator/blob/master/deployments/gpu-operator/values.yaml) -# You may use the NVIDIA GPU Operator to manage installation of NVIDIA drivers and the container toolkit on nodes with attached GPUs. -# @default -- `` -gpu-operator: - # -- Whether to install the NVIDIA GPU Operator to manage driver and/or container toolkit installation. - # See the list of [supported Operating Systems](https://docs.nvidia.com/datacenter/cloud-native/gpu-operator/latest/platform-support.html#supported-operating-systems-and-kubernetes-platforms) - # to verify compatibility with your cluster/nodes. Disable this option if your cluster/nodes are not compatible. - # If disabled, you will need to self-manage NVIDIA software installation on all nodes where you want - # to schedule Deepgram Engine pods. - enabled: false - driver: - # -- Whether to install NVIDIA drivers on nodes where a NVIDIA GPU is detected. - # If your Kubernetes nodes run a base image that comes with NVIDIA drivers pre-configured, - # disable this option, but keep the parent `gpu-operator` and sibling `toolkit` - # options enabled. - enabled: true - # -- NVIDIA driver version to install. - version: "550.54.15" - toolkit: - # -- Whether to install NVIDIA drivers on nodes where a NVIDIA GPU is detected. - enabled: true - # -- NVIDIA container toolkit to install. The default `ubuntu` image tag for the - # toolkit requires a dynamic runtime link to a version of GLIBC that may not be - # present on nodes running older Linux distribution releases, such as Ubuntu 22.04. - # Therefore, we specify the `ubi8` image, which statically links the GLIBC library - # and avoids this issue. - version: v1.15.0-ubi8 - -cluster-autoscaler: - # -- Set to `true` to enable node autoscaling with AWS EKS. Note needed for GKE, as autoscaling is enabled by a - # [cli option on cluster creation](https://cloud.google.com/kubernetes-engine/docs/how-to/cluster-autoscaler#creating_a_cluster_with_autoscaling). - enabled: false - rbac: - serviceAccount: - # -- Name of the IAM Service Account with the [necessary permissions](https://github.com/kubernetes/autoscaler/blob/master/cluster-autoscaler/cloudprovider/aws/README.md#permissions) - name: cluster-autoscaler-sa - annotations: - # -- (string) Replace with the AWS Role ARN configured for the Cluster Autoscaler. - # See the [Deepgram AWS EKS guide](https://developers.deepgram.com/docs/aws-k8s#creating-a-cluster) - # or [Cluster Autoscaler AWS documentation](https://github.com/kubernetes/autoscaler/blob/master/cluster-autoscaler/cloudprovider/aws/README.md#permissions) - # for details. - eks.amazonaws.com/role-arn: - autoDiscovery: - # -- (string) Name of your AWS EKS cluster. Using the [Cluster Autoscaler](https://github.com/kubernetes/autoscaler) - # on AWS requires knowledge of certain cluster metadata. - clusterName: - # -- (string) Region of your AWS EKS cluster. Using the [Cluster Autoscaler](https://github.com/kubernetes/autoscaler) - # on AWS requires knowledge of certain cluster metadata. - awsRegion: - -# -- Passthrough values for [Prometheus k8s stack Helm chart](https://github.com/prometheus-community/helm-charts/tree/main/charts/kube-prometheus-stack). -# Prometheus (and its adapter) should be configured when scaling.auto is enabled. -# You may choose to use the installation/configuration bundled in this Helm chart, -# or you may configure an existing Prometheus installation in your cluster to expose -# the needed values. -# See source Helm chart for explanation of available values. Default values provided in this chart are used -# to provide pod autoscaling for Deepgram pods. -# @default -- `` -kube-prometheus-stack: - # -- (bool) Normally, this chart will be installed if `scaling.auto.enabled` is true. However, if you wish - # to manage the Prometheus adapter in your cluster on your own and not as part of the Deepgram Helm chart, - # you can force it to not be installed by setting this to `false`. - includeDependency: false - - fullnameOverride: "dg-prometheus-stack" - prometheus: - prometheusSpec: - additionalScrapeConfigs: - - job_name: "dg_engine_metrics" - scrape_interval: "2s" - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - "{{ .Release.Namespace }}" - relabel_configs: - - source_labels: [__meta_kubernetes_service_name] - regex: "(.*)-metrics" - action: keep - - source_labels: [__meta_kubernetes_endpoint_port_name] - regex: "metrics" - action: keep - - source_labels: [__meta_kubernetes_namespace] - target_label: namespace - - source_labels: [__meta_kubernetes_service_name] - target_label: service - - source_labels: [__meta_kubernetes_pod_name] - target_label: pod - - prometheusOperator: - enabled: false - - alertmanager: - enabled: false - - grafana: - enabled: false - - nodeExporter: - enabled: false - - kube-state-metrics: - enabled: false - metricLabelsAllowlist: - - namespaces=[{{ .Release.Namespace }}],deployments=[app] - -# -- Passthrough values for [Prometheus Adapter Helm chart](https://github.com/prometheus-community/helm-charts/tree/main/charts/prometheus-adapter). -# Prometheus, and its adapter here, should be configured when scaling.auto is enabled. -# You may choose to use the installation/configuration bundled in this Helm chart, -# or you may configure an existing Prometheus installation in your cluster to expose -# the needed values. -# See source Helm chart for explanation of available values. Default values provided in this chart are used -# to provide pod autoscaling for Deepgram pods. -# @default -- `` -prometheus-adapter: - # -- Normally, this chart will be installed if `scaling.auto.enabled` is true. However, if you wish - # to manage the Prometheus adapter in your cluster on your own and not as part of the Deepgram Helm chart, - # you can force it to not be installed by setting this to `false`. - includeDependency: - prometheus: - url: http://dg-prometheus-stack-prometheus.{{ .Release.Namespace }}.svc - rules: - default: false - external: - - name: - as: "engine_active_requests_stt_streaming" - seriesQuery: 'engine_active_requests{kind="stream"}' - metricsQuery: 'avg(engine_active_requests{kind="stream"})' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_active_requests_stt_batch" - seriesQuery: 'engine_active_requests{kind="batch"}' - metricsQuery: 'avg(engine_active_requests{kind="batch"})' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_active_requests_tts_batch" - seriesQuery: 'engine_active_requests{kind="tts"}' - metricsQuery: 'avg(engine_active_requests{kind="tts"})' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_estimated_stream_capacity" - seriesQuery: 'engine_active_requests{kind="stream"}' - metricsQuery: 'avg_over_time((sum(engine_active_requests{kind="stream"}) / sum(engine_estimated_stream_capacity) * 100)[1m:1m])' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_requests_active_to_max_ratio" - seriesQuery: "engine_max_active_requests" - metricsQuery: "avg_over_time((sum(engine_active_requests) / sum(engine_max_active_requests) * 100)[1m:1m])" - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_to_api_pod_ratio" - seriesQuery: 'kube_deployment_labels{label_app="deepgram-engine"}' - metricsQuery: '(sum(kube_deployment_status_replicas and on(deployment) kube_deployment_labels{label_app="deepgram-engine"})) / (sum(kube_deployment_status_replicas and on(deployment) kube_deployment_labels{label_app="deepgram-api"}))' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } diff --git a/helm-overrides/gke-supply-prd-ase1a/external-secrets/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/external-secrets/custom-values.yaml deleted file mode 100644 index 9bdb22a..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/external-secrets/custom-values.yaml +++ /dev/null @@ -1,61 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Annotations to add to Pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: supply-devops - nodeSelector: - cloud.google.com/compute-class: supply-devops - certController: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: supply-devops - nodeSelector: - cloud.google.com/compute-class: supply-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - webhook: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: supply-devops - nodeSelector: - cloud.google.com/compute-class: supply-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" \ No newline at end of file diff --git a/helm-overrides/gke-supply-prd-ase1a/flagger/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/flagger/custom-values.yaml deleted file mode 100644 index b3d3540..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/flagger/custom-values.yaml +++ /dev/null @@ -1,63 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - cloud.google.com/compute-class: supply-devops - -tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: supply-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: supply-devops - bu: infra - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-supply-rollout-service.prd-supply-rollout-service.svc.cluster.local" - -rolloutService: - enabled: true - apps: - \ No newline at end of file diff --git a/helm-overrides/gke-supply-prd-ase1a/fluentd/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/fluentd/custom-values.yaml deleted file mode 100644 index eb6f912..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/fluentd/custom-values.yaml +++ /dev/null @@ -1,854 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-supl-spsre-fluentd-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: cloud.google.com/compute-class - operator: NotIn - values: - - no-exclusion -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: supply - team: sre - type: fluentd - service: fluentd-supply-a-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: supply - team: sre - type: fluentd - service: fluentd-supply-a-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -# - secretRef: -# name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: - - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - from_encoding "utf-8" - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @DG_SELF_HOSTED - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/gke-supply-prd-ase1a/keda/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/keda/custom-values.yaml deleted file mode 100644 index 551321b..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/keda/custom-values.yaml +++ /dev/null @@ -1,45 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - cloud.google.com/compute-class: supply-devops - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "supply-devops" - effect: "NoSchedule" - podLabels: - bu: "supply" - team: "supply-devops" - metricsAdapter: - bu: "supply" - team: "supply-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 2000Mi - requests: - cpu: 700m - memory: 1000Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 120m - memory: 300Mi - metricServer: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 120m - memory: 200Mi - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' diff --git a/helm-overrides/gke-supply-prd-ase1a/kube-dns/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/kube-dns/custom-values.yaml deleted file mode 100644 index b6b5357..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.1.16.2"],"prd.mrouter.int.svc.cluster.local":["10.1.16.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.1.16.2"]} diff --git a/helm-overrides/gke-supply-prd-ase1a/kube-events/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/kube-events/custom-values.yaml deleted file mode 100644 index 635d304..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-supply-a-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - cloud.google.com/compute-class: supply-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: supply-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: supply-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: supply-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: supply-devops - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: supply-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/gke-supply-prd-ase1a/kube-state-metrics/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/kube-state-metrics/custom-values.yaml deleted file mode 100644 index d91c5e4..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-supply-a-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-supply-a-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "supply" - team: "supply-sre" - service: "kube-state-metrics-supply-a-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 350m - memory: 500Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/gke-supply-prd-ase1a/kubectl-mcp-server/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/kubectl-mcp-server/custom-values.yaml deleted file mode 100644 index 7e32ed5..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/kubectl-mcp-server/custom-values.yaml +++ /dev/null @@ -1,117 +0,0 @@ -fullnameOverride: "kubectl-mcp-server" - -replicas: 1 - -image: - # In-house hardened build (helm-templates/kubectl-mcp-server/docker/) — - # NOT the upstream Docker Hub image. 266→18 HIGH/CRIT vulns, non-root, - # multi-stage slim base, kubectl v1.33.12 / helm v3.21.0. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/prd/devop/kubectl-mcp-server - tag: "v2" - pullPolicy: IfNotPresent - -labels: - bu: infra - team: devops - service: kubectl-mcp-server - env: prd - -serviceAccount: - create: true - annotations: {} - -rbac: - create: true - -podAnnotations: {} - -podSecurityContext: {} - -securityContext: - runAsNonRoot: true - runAsUser: 1000 - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -priorityClassName: "" - -nodeSelector: - cloud.google.com/compute-class: "supply-devops" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "supply-devops" - effect: NoSchedule - -affinity: {} - -resources: - requests: - cpu: 250m - memory: 512Mi - limits: - cpu: 1000m - memory: 1Gi - -mcp: - mode: single - transport: http - host: "0.0.0.0" - port: 8000 - -auth: - allowAnonymous: false - -externalSecrets: - enabled: true - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/prd/cntr/devop/kubectl-mcp-server-supply" - -# tcpSocket on purpose — streamable-http transport exposes only /mcp, -# there is no /health route (see chart values.yaml note). -livenessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - -readinessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - -service: - port: 8000 - type: ClusterIP - -# Contour standard (2.0.0-style): renders HTTPProxy parent/child + -intra -# instead of a networking.k8s.io Ingress. -createContourGateway: true - -ingress: - enabled: true - ingressClassName: contour-internal-1 - servicePortNumber: 8000 - hosts: - - host: kubectl-mcp-server-supply.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 diff --git a/helm-overrides/gke-supply-prd-ase1a/kubernetes-dashboard/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/kubernetes-dashboard/custom-values.yaml deleted file mode 100644 index 670d841..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/kubernetes-dashboard/custom-values.yaml +++ /dev/null @@ -1,442 +0,0 @@ -# Copyright 2017 The Kubernetes Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -# General configuration shared across resources -app: - # Mode determines if chart should deploy a full Dashboard with all containers or just the API. - # - dashboard - deploys all the containers - # - api - deploys just the API - mode: 'dashboard' - image: - pullPolicy: IfNotPresent - pullSecrets: [] - scheduling: - # Node labels for pod assignment - # Ref: https://kubernetes.io/docs/user-guide/node-selection/ - nodeSelector: {} - security: - # Allow overriding csrfKey used by API/Auth containers. - # It has to be base64 encoded random 256 bytes string. - # If empty, it will be autogenerated. - csrfKey: ~ - # SecurityContext to be added to pods - # To disable set the following configuration to null: - # securityContext: null - securityContext: - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - # ContainerSecurityContext to be added to containers - # To disable set the following configuration to null: - # containerSecurityContext: null - containerSecurityContext: - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - runAsUser: 1001 - runAsGroup: 2001 - capabilities: - drop: ["ALL"] - # Pod Disruption Budget configuration - # Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - minAvailable: 0 - maxUnavailable: 0 - networkPolicy: - enabled: false - ingressDenyAll: false - # Raw network policy spec that overrides predefined spec - # Example: - # spec: - # egress: - # - ports: - # - port: 123 - spec: {} - - # Common labels & annotations shared across all deployed resources - labels: {} - annotations: {} - # Common priority class used for all deployed resources - priorityClassName: null - settings: - ## Global dashboard settings - global: - # # Cluster name that appears in the browser window title if it is set - # clusterName: "" - # # Max number of items that can be displayed on each list page - # itemsPerPage: 10 - # # Max number of labels that are displayed by default on most views. - # labelsLimit: 3 - # # Number of seconds between every auto-refresh of logs - # logsAutoRefreshTimeInterval: 5 - # # Number of seconds between every auto-refresh of every resource. Set 0 to disable - # resourceAutoRefreshTimeInterval: 10 - # # Hide all access denied warnings in the notification panel - # disableAccessDeniedNotifications: false - # # Hide all namespaces option in namespace selection dropdown to avoid accidental selection in large clusters thus preventing OOM errors - # hideAllNamespaces: false - # # Namespace that should be selected by default after logging in. - # defaultNamespace: default - # # List of namespaces that should be presented to user without namespace list privileges. - # namespaceFallbackList: - # - default - ## Pinned resources that will be displayed in dashboard's menu - pinnedResources: [] - # - kind: customresourcedefinition - # # Fully qualified name of a CRD - # name: prometheus.monitoring.coreos.com - # # Display name - # displayName: Prometheus - # # Is this CRD namespaced? - # namespaced: true - ingress: - enabled: false - hosts: - # Keep 'localhost' host only if you want to access Dashboard using 'kubectl port-forward ...' on: - # https://localhost:8443 - - localhost - # - kubernetes.dashboard.domain.com - ingressClassName: internal-nginx - # Use only if your ingress controllers support default ingress classes. - # If set to true ingressClassName will be ignored and not added to the Ingress resources. - # It should fall back to using IngressClass marked as the default. - useDefaultIngressClass: false - # This will append our Ingress with annotations required by our default configuration. - # nginx.ingress.kubernetes.io/backend-protocol: "HTTPS" - # nginx.ingress.kubernetes.io/ssl-passthrough: "true" - # nginx.ingress.kubernetes.io/ssl-redirect: "true" - useDefaultAnnotations: true - pathType: ImplementationSpecific - # If path is not the default (/), rewrite-target annotation will be added to the Ingress. - # It allows serving Kubernetes Dashboard on a sub-path. Make sure that the configured path - # does not conflict with gateway route configuration. - path: / - issuer: - name: selfsigned - # Scope determines what kind of issuer annotation will be used on ingress resource - # - default - adds 'cert-manager.io/issuer' - # - cluster - adds 'cert-manager.io/cluster-issuer' - # - disabled - disables cert-manager annotations - scope: default - tls: - enabled: true - # If provided it will override autogenerated secret name - secretName: "" - labels: {} - annotations: {} - # Use the following toleration if Dashboard can be deployed on a tainted control-plane nodes - # - key: node-role.kubernetes.io/control-plane - # effect: NoSchedule - tolerations: [] - affinity: {} - -auth: - role: auth - image: - repository: docker.io/kubernetesui/dashboard-auth - tag: 1.2.2 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: auth - containerPort: 8000 - protocol: TCP - args: [] - env: [] - volumeMounts: - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for Auth related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -# API deployment configuration -api: - role: api - image: - repository: docker.io/kubernetesui/dashboard-api - tag: 1.10.1 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: api - containerPort: 8000 - protocol: TCP - # Additional container arguments - # Full list of arguments: https://github.com/kubernetes/dashboard/blob/master/docs/common/arguments.md - # args: - # - --system-banner="Welcome to the Kubernetes Dashboard" - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store exec logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for API related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -# WEB UI deployment configuration -web: - role: web - image: - repository: docker.io/kubernetesui/dashboard-web - tag: 1.6.0 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: web - containerPort: 8000 - protocol: TCP - # Additional container arguments - # Full list of arguments: https://github.com/kubernetes/dashboard/blob/master/docs/common/arguments.md - # args: - # - --system-banner="Welcome to the Kubernetes Dashboard" - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for WEB UI related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -### Metrics Scraper -### Container to scrape, store, and retrieve a window of time from the Metrics Server. -### refs: https://github.com/kubernetes/dashboard/tree/master/modules/metrics-scraper -metricsScraper: - enabled: true - role: metrics-scraper - image: - repository: docker.io/kubernetesui/dashboard-metrics-scraper - tag: 1.2.1 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - containerPort: 8000 - protocol: TCP - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - livenessProbe: - httpGet: - scheme: HTTP - path: / - port: 8000 - initialDelaySeconds: 30 - timeoutSeconds: 30 - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for Metrics Scraper related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -## Optional Metrics Server sub-chart configuration -## Enable this if you don't already have metrics-server enabled on your cluster and -## want to use it with dashboard metrics-scraper -## refs: -## - https://github.com/kubernetes-sigs/metrics-server -## - https://github.com/kubernetes-sigs/metrics-server/tree/master/charts/metrics-server -metrics-server: - enabled: false - args: - - --kubelet-preferred-address-types=InternalIP - - --kubelet-insecure-tls - -## Required Kong sub-chart with DBless configuration to act as a gateway -## for our all containers. -kong: - enabled: true - admin: - tls: - enabled: false - ## Configuration reference: https://docs.konghq.com/gateway/3.6.x/reference/configuration - env: - dns_order: LAST,A,CNAME,AAAA,SRV - plugins: 'off' - nginx_worker_processes: 1 - ingressController: - enabled: false - manager: - enabled: false - dblessConfig: - configMap: kong-dbless-config - proxy: - type: ClusterIP - http: - enabled: true - -## Optional Cert Manager sub-chart configuration -## Enable this if you don't already have cert-manager enabled on your cluster. -cert-manager: - enabled: false - installCRDs: true - -## Optional Nginx Ingress sub-chart configuration -## Enable this if you don't already have nginx-ingress enabled on your cluster. -nginx: - enabled: false - controller: - electionID: ingress-controller-leader - ingressClassResource: - name: internal-nginx - default: false - controllerValue: k8s.io/internal-ingress-nginx - service: - type: ClusterIP - -## Extra configurations: -## - manifests -## - predefined roles -## - prometheus -## - etc... -extras: - # Extra Kubernetes manifests to be deployed - # manifests: - # - apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: additional-configmap - # data: - # mykey: myvalue - manifests: [] - serviceMonitor: - # Whether to create a Prometheus Operator service monitor. - enabled: false - # Here labels can be added to the serviceMonitor - labels: {} - # Here annotations can be added to the serviceMonitor - annotations: {} - # metrics.serviceMonitor.metricRelabelings Specify Metric Relabelings to add to the scrape endpoint - # ref: https://github.com/coreos/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # metrics.serviceMonitor.relabelings [array] Prometheus relabeling rules - relabelings: [] - # ServiceMonitor connection scheme. Defaults to HTTPS. - scheme: https - # ServiceMonitor connection tlsConfig. Defaults to {insecureSkipVerify:true}. - tlsConfig: - insecureSkipVerify: true diff --git a/helm-overrides/gke-supply-prd-ase1a/kyverno/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/kyverno/custom-values.yaml deleted file mode 100644 index 53434f5..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/kyverno/custom-values.yaml +++ /dev/null @@ -1,2244 +0,0 @@ -global: - - # -- Internal settings used with `helm template` to generate install manifest - # @ignored - templating: - enabled: false - debug: false - version: ~ - - image: - # -- (string) Global value that allows to set a single image registry across all deployments. - # When set, it will override any values set under `.image.registry` across the chart. - registry: ~ - # -- (list) Global list of Image pull secrets - # When set, it will override any values set under `imagePullSecrets` under different components across the chart. - imagePullSecrets: [] - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - caCertificates: - # -- Global CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - # Individual controller values will override this global value - data: ~ - - # -- Global value to set single volume to be mounted for CA certificates for all deployments. - # Not used when `.Values.global.caCertificates.data` is defined - # Individual controller values will override this global value - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Additional container environment variables to apply to all containers and init containers - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Global node labels for pod assignment. Non-global values will override the global value. - nodeSelector: - cloud.google.com/compute-class: supply-kyverno - - # -- Global List of node taints to tolerate. Non-global values will override the global value. - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "supply-kyverno" - effect: "NoSchedule" - -# -- (string) Override the name of the chart -nameOverride: ~ - -# -- (string) Override the expanded name of the chart -fullnameOverride: ~ - -# -- (string) Override the namespace the chart deploys to -namespaceOverride: ~ - -upgrade: - # -- Upgrading from v2 to v3 is not allowed by default, set this to true once changes have been reviewed. - fromV2: false - -apiVersionOverride: - # -- (string) Override api version used to create `PodDisruptionBudget`` resources. - # When not specified the chart will check if `policy/v1/PodDisruptionBudget` is available to - # determine the api version automatically. - podDisruptionBudget: ~ - -rbac: - roles: - # -- Aggregate ClusterRoles to Kubernetes default user-facing roles. For more information, see [User-facing roles](https://kubernetes.io/docs/reference/access-authn-authz/rbac/#user-facing-roles) - aggregate: - admin: true - view: true - -# Use openreports.io as the API group for reporting -openreports: - # -- Enable OpenReports feature in controllers - enabled: false - # -- Whether to install CRDs from the upstream OpenReports chart. Setting this to true requires enabled to also be true. - installCrds: false - -# CRDs configuration -crds: - - # -- Whether to have Helm install the Kyverno CRDs, if the CRDs are not installed by Helm, they must be added before policies can be created - install: true - - reportsServer: - # -- Kyverno reports-server is used in your cluster - enabled: false - - groups: - - # -- Install CRDs in group `kyverno.io` - kyverno: - cleanuppolicies: true - clustercleanuppolicies: true - clusterpolicies: true - globalcontextentries: true - policies: true - policyexceptions: true - updaterequests: true - - # -- Install CRDs in group `policies.kyverno.io` - policies: - validatingpolicies: true - policyexceptions: true - imagevalidatingpolicies: true - namespacedimagevalidatingpolicies: true - mutatingpolicies: true - generatingpolicies: true - deletingpolicies: true - namespaceddeletingpolicies: true - namespacedvalidatingpolicies: true - - # -- Install CRDs in group `reports.kyverno.io` - reports: - clusterephemeralreports: true - ephemeralreports: true - - # -- Install CRDs in group `wgpolicyk8s.io` - wgpolicyk8s: - clusterpolicyreports: false ##lk - policyreports: false ##lk - - # -- Additional CRDs annotations (Replace=true avoids 262144-byte annotation limit on large CRDs) - annotations: - argocd.argoproj.io/sync-options: Replace=true - # strategy.spinnaker.io/replace: 'true' - - # -- Additional CRDs labels - customLabels: {} - - migration: - - # -- Enable CRDs migration using helm post upgrade hook - enabled: true - - # -- Resources to migrate - resources: - - cleanuppolicies.kyverno.io - - clustercleanuppolicies.kyverno.io - - clusterpolicies.kyverno.io - - globalcontextentries.kyverno.io - - policies.kyverno.io - - policyexceptions.kyverno.io - - updaterequests.kyverno.io - - deletingpolicies.policies.kyverno.io - - generatingpolicies.policies.kyverno.io - - imagevalidatingpolicies.policies.kyverno.io - - namespacedimagevalidatingpolicies.policies.kyverno.io - - mutatingpolicies.policies.kyverno.io - - namespaceddeletingpolicies.policies.kyverno.io - - namespacedvalidatingpolicies.policies.kyverno.io - - policyexceptions.policies.kyverno.io - - validatingpolicies.policies.kyverno.io - - image: - # -- (string) Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- (string) Image repository - repository: kyverno/kyverno-cli - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- (string) Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podResources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -# Configuration -config: - - # -- Create the configmap. - create: true - - # -- Preserve the configmap settings during upgrade. - preserve: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - # -- Enable registry mutation for container images. Enabled by default. - enableDefaultRegistryMutation: true - - # -- The registry hostname used for the image mutation. - defaultRegistry: docker.io - - # -- Exclude groups - excludeGroups: - - system:nodes - - # -- Exclude usernames - excludeUsernames: [] - # - '!system:kube-scheduler' - - # -- Exclude roles - excludeRoles: [] - - # -- Exclude roles - excludeClusterRoles: [] - - # -- Generate success events. - generateSuccessEvents: false - - # -- Resource types to be skipped by the Kyverno policy engine. - # Make sure to surround each entry in quotes so that it doesn't get parsed as a nested YAML list. - # These are joined together without spaces, run through `tpl`, and the result is set in the config map. - # @default -- See [values.yaml](values.yaml) - resourceFilters: - - '[Event,*,*]' - # - '[*/*,kube-system,*]' - - '[*/*,kube-public,*]' - - '[*/*,kube-node-lease,*]' - - '[Node,*,*]' - - '[Node/?*,*,*]' - - '[APIService,*,*]' - - '[APIService/?*,*,*]' - - '[TokenReview,*,*]' - - '[SubjectAccessReview,*,*]' - - '[SelfSubjectAccessReview,*,*]' - - '[Binding,*,*]' - - '[Pod/binding,*,*]' - - '[ReplicaSet,*,*]' - - '[ReplicaSet/?*,*,*]' - - '[EphemeralReport,*,*]' - - '[ClusterEphemeralReport,*,*]' - # exclude resources from the chart - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:additional]' - - '[ClusterRoleBinding,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.configMapName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.metricsConfigMapName" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Job,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[Job/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.admission-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.background-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - # -- Sets the threshold for the total number of UpdateRequests generated for mutateExisitng and generate policies. - updateRequestThreshold: 1000 - - # -- Defines the `namespaceSelector`/`objectSelector` in the webhook configurations. - # The Kyverno namespace is excluded if `excludeKyvernoNamespace` is `true` (default) - webhooks: - # Exclude namespaces - # namespaceSelector: lk-s2 - # matchExpressions: - # - key: kubernetes.io/metadata.name - # operator: NotIn - # values: - # - kube-systemm lk-s2 - # Exclude objects - # objectSelector: - # matchExpressions: - # - key: webhooks.kyverno.io/exclude - # operator: DoesNotExist - - # -- Defines annotations to set on webhook configurations. - webhookAnnotations: - # Example to disable admission enforcer on AKS: - # 'admissions.enforcer/disabled': 'true' lk-s3 - - # -- Defines labels to set on webhook configurations. - webhookLabels: {} - # Example to adopt webhook resources in ArgoCD: - # 'argocd.argoproj.io/instance': 'kyverno' - - # -- Defines match conditions to set on webhook configurations (requires Kubernetes 1.27+). - matchConditions: [] - - # -- Exclude Kyverno namespace - # Determines if default Kyverno namespace exclusion is enabled for webhooks and resourceFilters - excludeKyvernoNamespace: true - - # -- resourceFilter namespace exclude - # Namespaces to exclude from the default resourceFilters - resourceFiltersExcludeNamespaces: [] - - # -- resourceFilters exclude list - # Items to exclude from config.resourceFilters - resourceFiltersExclude: [] - - # -- resourceFilter namespace include - # Namespaces to include to the default resourceFilters - resourceFiltersIncludeNamespaces: [] - - # -- resourceFilters include list - # Items to include to config.resourceFilters - resourceFiltersInclude: [] - -# Metrics configuration -metricsConfig: - - # -- Create the configmap. - create: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - namespaces: - - # -- List of namespaces to capture metrics for. - include: [] - - # -- list of namespaces to NOT capture metrics for. - exclude: [] - - # -- (string) Rate at which metrics should reset so as to clean up the memory footprint of kyverno metrics, if you might be expecting high memory footprint of Kyverno's metrics. Default: 0, no refresh of metrics. WARNING: This flag is not working since Kyverno 1.8.0 - metricsRefreshInterval: ~ - # metricsRefreshInterval: 24h - - # -- (list) Configures the bucket boundaries for all Histogram metrics, changing this configuration requires restart of the kyverno admission controller - bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 15, 20, 25, 30] - - # -- (map) Configures the exposure of individual metrics, by default all metrics and all labels are exported, changing this configuration requires restart of the kyverno admission controller - metricsExposure: - kyverno_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_image_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_mutating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_generating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_admission_review_duration_seconds: - # enabled: false - disabledLabelDimensions: ["resource_namespace"] - kyverno_policy_rule_info_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_policy_results_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_admission_requests_total: - disabledLabelDimensions: ["resource_namespace"] - kyverno_cleanup_controller_deletedobjects_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - -# -- Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -imagePullSecrets: {} - # regcred: - # registry: foo.example.com - # username: foobar - # password: secret - # regcred2: - # registry: bar.example.com - # username: barbaz - # password: secret2 - -# -- Existing Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -existingImagePullSecrets: [] - # - test-registry - # - other-test-registry - -# Tests configuration -test: - # -- Sleep time before running test - sleep: 20 - - image: - # -- (string) Image registry - registry: curlimages - # -- Image repository - repository: curl - # -- Image tag - # Defaults to `latest` if omitted - tag: '8.10.1' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - # -- Security context for the test containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- Additional Pod annotations - podAnnotations: {} - - # -- List of node taints to tolerate - tolerations: [] - -# -- Additional labels -customLabels: {} - - -webhooksCleanup: - # -- Create a helm pre-delete hook to cleanup webhooks. - enabled: true - - autoDeleteWebhooks: - # -- Allow webhooks controller to delete webhooks using finalizers - enabled: false - - image: - # -- (string) Image registry - registry: registry.k8s.io - # -- Image repository - repository: kubectl - # -- Image tag - # Defaults to `latest` if omitted - tag: 'v1.32.7' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -grafana: - # -- Enable grafana dashboard creation. - enabled: false - - # -- Configmap name template. - configMapName: '{{ include "kyverno.fullname" . }}-grafana' - - # -- (string) Namespace to create the grafana dashboard configmap. - # If not set, it will be created in the same namespace where the chart is deployed. - namespace: ~ - - # -- Grafana dashboard configmap annotations. - annotations: {} - - # -- Grafana dashboard configmap labels - labels: - grafana_dashboard: "1" - - # -- create GrafanaDashboard custom resource referencing to the configMap. - # according to https://grafana-operator.github.io/grafana-operator/docs/examples/dashboard_from_configmap/readme/ - grafanaDashboard: - create: false - folder: kyverno - allowCrossNamespaceImport: true - matchLabels: - dashboards: "grafana" - -# Features configuration -features: - admissionReports: - # -- Enables the feature - enabled: true - aggregateReports: - # -- Enables the feature - enabled: true - policyReports: - # -- Enables the feature - enabled: false ##lk - validatingAdmissionPolicyReports: - # -- Enables the feature - enabled: true - mutatingAdmissionPolicyReports: - # -- Enables the feature - enabled: false - reporting: - # -- Enables the feature - validate: true - # -- Enables the feature - mutate: true - # -- Enables the feature - mutateExisting: true - # -- Enables the feature - imageVerify: true - # -- Enables the feature - generate: true - autoUpdateWebhooks: - # -- Enables the feature - enabled: true - backgroundScan: - # -- Enables the feature - enabled: true - # -- Number of background scan workers - backgroundScanWorkers: 2 - # -- Background scan interval - backgroundScanInterval: 1h - # -- Skips resource filters in background scan - skipResourceFilters: true - configMapCaching: - # -- Enables the feature - enabled: true - controllerRuntimeMetrics: - # -- Bind address for controller-runtime metrics (use "0" to disable it) - bindAddress: ":8080" - deferredLoading: - # -- Enables the feature - enabled: true - dumpPayload: - # -- Enables the feature - enabled: false - forceFailurePolicyIgnore: - # -- Enables the feature - enabled: false - generateValidatingAdmissionPolicy: - # -- Enables the feature - enabled: true - generateMutatingAdmissionPolicy: - # -- Enables the feature - enabled: false - dumpPatches: - # -- Enables the feature - enabled: false - globalContext: - # -- Maximum allowed response size from API Calls. A value of 0 bypasses checks (not recommended) - maxApiCallResponseLength: 2000000 - logging: - # -- Logging format - format: text - # -- Logging verbosity - verbosity: 2 - omitEvents: - # -- Events which should not be emitted (possible values `PolicyViolation`, `PolicyApplied`, `PolicyError`, and `PolicySkipped`) - eventTypes: - - PolicyApplied - - PolicySkipped - # - PolicyViolation - # - PolicyError - policyExceptions: - # -- Enables the feature - enabled: true - # -- Restrict policy exceptions to a single namespace - # Set to "*" to allow exceptions in all namespaces - namespace: '' - protectManagedResources: - # -- Enables the feature - enabled: false - registryClient: - # -- Allow insecure registry - allowInsecure: false - # -- Enable registry client helpers - credentialHelpers: - - default - - google - - amazon - - azure - - github - ttlController: - # -- Reconciliation interval for the label based cleanup manager - reconciliationInterval: 1m - tuf: - # -- Enables the feature - enabled: false - # -- (string) Path to Tuf root - root: ~ - # -- (string) Raw Tuf root - rootRaw: ~ - # -- (string) Tuf mirror - mirror: ~ - -# Admission controller configuration -admissionController: - autoscaling: - # -- Enable horizontal pod autoscaling - enabled: true - - # -- Minimum number of pods - minReplicas: 1 - - # -- Maximum number of pods - maxReplicas: 10 - - # -- Target CPU utilization percentage - targetCPUUtilizationPercentage: 80 - - # -- Configurable scaling behavior - behavior: {} - - # -- Overrides features defined at the root level - featuresOverride: - admissionReports: - # -- Max number of admission reports allowed in flight until the admission controller stops creating new ones - backPressureThreshold: 1000 - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- The ServiceAccount name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno admission controller activities. - # This will help ensure Kyverno stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- admissionController webhook server port - # in case you are using hostNetwork: true, you might want to change the port the webhookServer is listening to - webhookServer: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - admission-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.admissionController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - initContainer: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyvernopre - # -- (string) Image tag - # If missing, defaults to image.tag - tag: ~ - # -- (string) Image pull policy - # If missing, defaults to image.pullPolicy - pullPolicy: ~ - - resources: - # -- Pod resource limits - limits: - cpu: 1000m - memory: 1Gi - # -- Pod resource requests - requests: - cpu: 1000m - memory: 1Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - container: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyverno - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - resources: - # -- Pod resource limits - limits: - cpu: 1900m - memory: 3Gi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Array of extra init containers - extraInitContainers: [] - # - name: init-container - # image: busybox - # command: ['sh', '-c', 'echo Hello'] - - # -- Array of extra containers to run alongside kyverno - extraContainers: [] - # - name: myapp-container - # image: busybox - # command: ['sh', '-c', 'echo Hello && sleep 3600'] - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Kyverno's metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Background controller configuration -backgroundController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable background controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: - - apiGroups: - - networking.k8s.io - resources: - - ingresses - - ingressclasses - - networkpolicies - verbs: - - create - - update - - patch - - delete - - apiGroups: - - rbac.authorization.k8s.io - resources: - - rolebindings - - roles - verbs: - - create - - update - - patch - - delete - - apiGroups: - - '' - resources: - - configmaps - - resourcequotas - - limitranges - verbs: - - create - - update - - patch - - delete - - apiGroups: - - resource.k8s.io - resources: - - resourceclaims - - resourceclaimtemplates - verbs: - - create - - delete - - update - - patch - - deletecollection - - apiGroups: - - autoscaling - resources: - - horizontalpodautoscalers - verbs: - - get - - list - - watch - - update - - patch - - apiGroups: - - apps - resources: - - deployments - - statefulsets - - replicasets - verbs: - - get - - list - - watch - - update - - patch - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - # - patch - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/background-controller - # -- Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 1900m - memory: 3Gi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - background-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.backgroundController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - # -- backgroundController server port - # in case you are using hostNetwork: true, you might want to change the port the backgroundController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Cleanup controller configuration -cleanupController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable cleanup controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - delete - # - list - # - watch - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/cleanup-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- cleanupController server port - # in case you are using hostNetwork: true, you might want to change the port the cleanupController is listening to - server: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 1100m - memory: 1.2Gi - # -- Pod resource requests - requests: - cpu: 1100m - memory: 1.2Gi - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - cleanup-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `service.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Reports controller configuration -reportsController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable reports controller. - enabled: false ##lk - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/reports-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno reports controller activities. - # This will help ensure Kyverno reports stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - cpu: 200m - memory: 128Mi - # -- Pod resource requests - requests: - cpu: 100m - memory: 64Mi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - reports-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.reportsController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- (string) Service node port. - # Only used if `type` is `NodePort`. - nodePort: ~ - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- (string) Traces receiver address - address: ~ - # -- (string) Traces receiver port - port: ~ - # -- (string) Traces receiver credentials - creds: ~ - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- (string) Otel collector endpoint - collector: ~ - # -- (string) Otel collector credentials - creds: ~ - - # -- reportsController server port - # in case you are using hostNetwork: true, you might want to change the port the reportsController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - - # -- Enable sanity check for reports CRDs - sanityChecks: true diff --git a/helm-overrides/gke-supply-prd-ase1a/kyverno/policies/protect-namespace.yaml b/helm-overrides/gke-supply-prd-ase1a/kyverno/policies/protect-namespace.yaml deleted file mode 100644 index c7f6c22..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/kyverno/policies/protect-namespace.yaml +++ /dev/null @@ -1,40 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: protect-critical-namespaces -spec: - validationFailureAction: Enforce - background: false - rules: - - name: deny-deletion-of-critical-namespaces - match: - any: - - resources: - kinds: - - Namespace - names: - - keda-supply-a-prd - - contour-internal-0-supply-a-prd - - contour-internal-0-supply-a-prd-intra - - contour-external-supply-a-prd - - external-secrets-supply-a-prd - - flagger-supply-a-prd - - victoriametrics - - kube-system - - monitoring - - telegraf-operator - - loadtester - - opentelemetry - - fluentd - - kube-events - - observability - - contour-cert-checker-ns - - aurva-dataplane - validate: - message: "Deletion of critical namespaces is not allowed." - deny: - conditions: - any: - - key: "{{request.operation}}" - operator: Equals - value: DELETE diff --git a/helm-overrides/gke-supply-prd-ase1a/kyverno/policies/restrict-replicas.yaml b/helm-overrides/gke-supply-prd-ase1a/kyverno/policies/restrict-replicas.yaml deleted file mode 100644 index d456985..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/kyverno/policies/restrict-replicas.yaml +++ /dev/null @@ -1,29 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: restrict-minimum-replicas-selected-ns -spec: - validationFailureAction: Enforce - background: false - rules: - - name: min-replicas-deployments-statefulsets - match: - resources: - kinds: - - Deployment - - StatefulSet - - Deployment/scale - - StatefulSet/scale - names: - - coredns - namespaces: - - kube-system - operations: - - CREATE - - UPDATE - validate: - message: "Deployments and StatefulSets must have at least 5 replicas in this namespace." - anyPattern: - # Case 1: replicas explicitly set and >= 5 - - spec: - replicas: ">=5" diff --git a/helm-overrides/gke-supply-prd-ase1a/loadtester/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/loadtester/custom-values.yaml deleted file mode 100644 index 059e021..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/loadtester/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -replicaCount: 1 - -Namespace: loadtester - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/dev/devops/flagger-loadtester - tag: 0.30.0 - pullPolicy: IfNotPresent - pullSecret: - -podLabels: - bu: supply - team: supply-shared - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - -podPriorityClassName: "" - -logLevel: info -cmd: - timeout: 1h - namespaceRegexp: "" - -nameOverride: "loadtester" -fullnameOverride: "" - -env: [] - -service: - type: ClusterIP - port: 80 - -resources: - requests: - cpu: 10m - memory: 64Mi - -volumes: [] -volumeMounts: [] - -nodeSelector: - cloud.google.com/compute-class: supply-devops - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "supply-devops" - effect: "NoSchedule" - -affinity: {} - -ingress: - slowStart: - enabled: false - enabled: true - ingressClassName: contour-internal-1 - servicePort: http - hosts: - - host: supply-flagger-loadtester.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / -createContourGateway: true -contourResponseTimeout: false -rbac: - # rbac.create: `true` if rbac resources should be created - create: true - # rbac.scope: `cluster` to create cluster-scope rbac resources (ClusterRole/ClusterRoleBinding) - # otherwise, namespace-scope rbac resources will be created (Role/RoleBinding) - scope: - # rbac.rules: array of rules to apply to the role. example: - # rules: - # - apiGroups: [""] - # resources: ["pods"] - # verbs: ["list", "get"] - rules: [] - -# name of an existing service account to use - if not creating rbac resources -serviceAccountName: "" - -# App Mesh virtual node settings (to be used for AppMesh v1beta1) -meshName: "" -#backends: -# - app1.namespace -# - app2.namespace - -# App Mesh virtual node settings (to be used for AppMesh v1beta2) -appmesh: - enabled: false - backends: - - podinfo - - podinfo-canary - -#Istio virtual service and gatway settings. TLS secrets should be in namespace before enbaled it. ( secret format loadtester.fullname ) -istio: - enabled: false - host: flagger-loadtester.flagger - gateway: - enabled: false - tls: - enabled: false - httpsRedirect: false - -# when enabled, it will add a security context for the loadtester pod -securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 100 - runAsGroup: 101 - -podDisruptionBudget: - enabled: false - minAvailable: 1 diff --git a/helm-overrides/gke-supply-prd-ase1a/node-thp-config/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/node-thp-config/custom-values.yaml deleted file mode 100644 index ec67538..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/node-thp-config/custom-values.yaml +++ /dev/null @@ -1,9 +0,0 @@ -daemonSet: - namespace: prd-node-thp-config - -baseMatchExpressions: -- key: dedicated - operator: In - values: - - "sumounolite-azul" # change to your actual node label value - - "megatetralite-azul" # adding megatetralite diff --git a/helm-overrides/gke-supply-prd-ase1a/opentelemetry-coralogix/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/opentelemetry-coralogix/custom-values.yaml deleted file mode 100644 index 5ed02ab..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/opentelemetry-coralogix/custom-values.yaml +++ /dev/null @@ -1,288 +0,0 @@ -global: - domain: "coralogixsg.com" - defaultApplicationName: "default" - defaultSubsystemName: "nodes" - - # Old endpoint based configuration, - # please use domain instead. - traces: - endpoint: "" - metrics: - endpoint: "" - logs: - endpoint: "" - -# set distribution to openshift for openshift clusters -distribution: "" - -# Opentelemetry collector configuration -dedicatedValue: false -schedulerName: default-scheduler - -labels: - bu: "supply" - team: "sre" - service: "opentelemetry-supply-a-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: false - key: prd/admin/coralogix-keys - secretStoreRef: - name: vault-backend - -image: - # If you want to use the core image `otel/opentelemetry-collector`, you also need to change `command.name` value to `otelcol`. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - # repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - tag: "0.87.0" - # When digest is set to a non-empty value, images will be pulled by digest (regardless of tag value). - digest: "" -mode: daemonset -rollout: - rollingUpdate: - # When 'mode: daemonset', maxSurge cannot be used when hostPort is set for any of the ports - # maxSurge: 25% - maxUnavailable: 15 - strategy: RollingUpdate -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" -hostNetwork: true -dnsPolicy: "ClusterFirstWithHostNet" -fullnameOverride: opentelemetry-supply-a-prd - -presets: - logsCollection: - enabled: false - storeCheckpoints: false - maxRecombineLogSize: 1048576 - extraFilelogOperators: [] -# - type: recombine -# combine_field: body -# source_identifier: attributes["log.file.path"] -# is_first_entry: body matches "^(YOUR-LOGS-REGEX)" - kubernetesAttributes: - enabled: false - hostMetrics: - enabled: false - kubeletMetrics: - enabled: false - -extraEnvs: -- name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" -- name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName -config: - extensions: {} - # zpages: - # endpoint: localhost:55679 - # pprof: - # endpoint: localhost:1777 - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 2s - tls: - insecure: true - sending_queue: - enabled: true - queue_size: 30000 - num_consumers: 500 - resolver: - dns: - hostname: opentelemetry-admin-prd.opentelemetry.svc.clusterset.local - # coralogix: - # timeout: "30s" - # private_key: "${CORALOGIX_PRIVATE_KEY}" - # domain: "{{.Values.global.domain}}" - # traces: - # endpoint: "{{ .Values.global.traces.endpoint }}" - # metrics: - # endpoint: "{{ .Values.global.metrics.endpoint }}" - # logs: - # endpoint: "{{ .Values.global.logs.endpoint }}" - # application_name_attributes: - # - "k8s.namespace.name" - # - "service.namespace" - # subsystem_name_attributes: - # - "k8s.deployment.name" - # - "k8s.statefulset.name" - # - "k8s.daemonset.name" - # - "k8s.cronjob.name" - # - "k8s.job.name" - # - "k8s.container.name" - # - "k8s.node.name" - # - "service.name" - # application_name: "{{.Values.global.defaultApplicationName }}" - # subsystem_name: "{{.Values.global.defaultSubsystemName }}" - processors: - k8sattributes: - filter: - node_from_env_var: KUBE_NODE_NAME - extract: - metadata: - - "k8s.namespace.name" - - "k8s.deployment.name" - - "k8s.statefulset.name" - - "k8s.daemonset.name" - - "k8s.cronjob.name" - - "k8s.job.name" - - "k8s.pod.name" - - "k8s.node.name" - memory_limiter: null # Will get the k8s resource limits - groupbytrace: - wait_duration: 1s - groupbyattrs: - keys: - - host.name - resourcedetection/env: - detectors: ["system","env"] - timeout: 5s - override: false - # spanmetrics: - # metrics_exporter: coralogix - # dimensions: - # - name: "k8s.deployment.name" - # - name: "k8s.statefulset.name" - # - name: "k8s.daemonset.name" - # - name: "k8s.cronjob.name" - # - name: "k8s.job.name" - # - name: "k8s.container.name" - # - name: "k8s.node.name" - # - name: "k8s.namespace.name" - receivers: - otlp: - protocols: - grpc: - endpoint: ${MY_POD_IP}:4317 - http: - endpoint: ${MY_POD_IP}:4318 - zipkin: - endpoint: ${MY_POD_IP}:9411 - jaeger: - protocols: - grpc: - endpoint: ${MY_POD_IP}:14250 - thrift_http: - endpoint: ${MY_POD_IP}:14268 - thrift_compact: - endpoint: ${MY_POD_IP}:6831 - thrift_binary: - endpoint: ${MY_POD_IP}:6832 - prometheus: - config: - scrape_configs: - - job_name: opentelemetry-collector - scrape_interval: 30s - static_configs: - - targets: - - ${MY_POD_IP}:8888 - service: - extensions: - # - zpages - # - pprof - - health_check -# - memory_ballast - telemetry: - metrics: - address: ${env:MY_POD_IP}:8888 - pipelines: - traces: - exporters: - - loadbalancing - processors: - - memory_limiter - - batch - receivers: - - otlp - metrics: null - logs: null -tolerations: - - operator: Exists - -resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 1 - memory: 2G - -ports: - jaeger-binary: - enabled: true - containerPort: 6832 - servicePort: 6832 - hostPort: 6832 - protocol: TCP - # In order to enable podMonitor, following part must be enabled in order to expose the required port: - # metrics: - # enabled: true - -# podMonitor: -# enabled: true - -# prometheusRule: -# enabled: true -# defaultRules: -# enabled: true - -# Annotations to be added to pod -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-supply-contour-ext - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-supply-default-prd-ase1 - - key: cloud.google.com/compute-class - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external diff --git a/helm-overrides/gke-supply-prd-ase1a/opentelemetry-daemonset-medium-np/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/opentelemetry-daemonset-medium-np/custom-values.yaml deleted file mode 100644 index fe33ae2..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/opentelemetry-daemonset-medium-np/custom-values.yaml +++ /dev/null @@ -1,110 +0,0 @@ -fullnameOverride: opentelemetry-medium-np-supply-a-prd - -mode: daemonset - -priorityClassName: "system-node-critical" - -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: In - values: - - sumoduolite-op - - key: dedicated - operator: In - values: - - sumounolite -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 350m - memory: 350Mi - limits: - cpu: 400m - memory: 512Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "supply" - team: "sre" - service: "opentelemetry-supply-a-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% diff --git a/helm-overrides/gke-supply-prd-ase1a/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index 9d1155c..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,138 +0,0 @@ -fullnameOverride: opentelemetry-supply-a-prd - -mode: daemonset - -priorityClassName: "system-node-critical" - -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-supply-contour-ext - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-supl-default-prd-ase1a - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external - - contour-shared - - contour-intra-0 - - contour-intra-1 - - alloy - - sumoduolite-op - - sumounolite -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 200m - memory: 256Mi - limits: - cpu: 400m - memory: 512Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "supply" - team: "sre" - service: "opentelemetry-supply-a-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% diff --git a/helm-overrides/gke-supply-prd-ase1a/prometheus-node-exporter/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index 11a53c0..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,496 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-supply-a-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "supply" - team: "supply-sre" - service: "node-exporter-supply-a-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/gke-supply-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index 67e3c5b..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-supply-a-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-supply-a-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "supply" - team: "supply-sre" - service: "stackdriver-exporter-supply-a-prd" - env: "prd" - priority: "p0" - type: "exporter" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-supply-prd-0622" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,compute.googleapis.com/instance,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-supl-spsre-stackdriver-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - } - - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/gke-supply-prd-ase1a/telegraf-operator-custom/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/telegraf-operator-custom/custom-values.yaml deleted file mode 100644 index 7ec27e5..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/telegraf-operator-custom/custom-values.yaml +++ /dev/null @@ -1,150 +0,0 @@ -replicaCount: 3 -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf-operator" - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "supply" - team: "supply-sre" - service: "telegraf-operator-custom-supply-a-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false diff --git a/helm-overrides/gke-supply-prd-ase1a/telegraf-operator/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/telegraf-operator/custom-values.yaml deleted file mode 100644 index c6c901e..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,232 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - - infra-histogram-optimized: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 80000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - namepass = ["DOWNSTREAM"] - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - namedrop = ["DOWNSTREAM"] - stats = ["count"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - expiration_interval = "1m" - [[aggregators.histogram.config]] - buckets = [10.0, 25.0, 50.0, 100.0, 250.0, 500.0, 1000.0, 2500.0, 5000.0, 10000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "supply" - team: "supply-sre" - service: "telegraf-operator-supply-a-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false \ No newline at end of file diff --git a/helm-overrides/gke-supply-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml deleted file mode 100644 index 5ddc0b5..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/victoria-metrics-agent-dr/custom-values.yaml +++ /dev/null @@ -1,295 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-supply-a-prd-dr -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-supply-a-prd-dr-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-supl-spsre-vmagent-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - - https://vminsert-infra-prd-dr.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "supply" - team: "supply-sre" - service: "vmagent-supply-a-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-supply-a-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-supply-a-prd-dr.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 12 - memory: 12Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - cloud.google.com/compute-class: "vmagent-dr" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/gke-supply-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml deleted file mode 100644 index 8465ba6..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/victoria-metrics-agent-fb/custom-values.yaml +++ /dev/null @@ -1,272 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 1 - -fullnameOverride: vmagent-supply-a-prd-fb -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-supply-a-prd-fb-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - -# Remove topology spread constraints -topologySpreadConstraints: [] - -# Modify deployment strategy -deployment: - enabled: true - -# Remove HPA -horizontalPodAutoscaler: - enabled: false - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-infr-sre-vmagent-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -# Remove pod disruption budget -podDisruptionBudget: - enabled: false - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-supply.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vmsinglenode-infra-prd.victoriametrics.svc.clusterset.local:8428/api/v1/write - - https://vmsinglenode-infra-prd.meeshogcp.in/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "supply-fb" - team: "sre" - service: "vmagent-supply-a-prd-fb" - env: "prd" - priority: "p0" - type: "vmagent-fb" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-supply-a-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-supply-a-prd-fb.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -# Modify resources -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - requests: - cpu: 2 - memory: 3Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - cloud.google.com/compute-class: "vmagent-n4" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-n4" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/gke-supply-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml deleted file mode 100644 index b3dcb88..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/victoria-metrics-alert-dr/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-supply-a-prd-dr - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/supply/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-supply-a-prd-dr.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-supply-a-prd-dr" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmagent-dr" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-supply-prd-ase1a/victoria-metrics-alert-secured-stateful/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/victoria-metrics-alert-secured-stateful/custom-values.yaml deleted file mode 100644 index cee4a6d..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/victoria-metrics-alert-secured-stateful/custom-values.yaml +++ /dev/null @@ -1,332 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-stateful-secured-supply-a-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-supply-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-supply-a-prd.victoriametrics.svc.cluster.local:8480/insert/200/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-supply-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/sensitive/supply/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: [] - # Extra Volume Mounts for the container - extraVolumeMounts: [] - extraContainers: [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: false - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-supply-a-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "supply" - team: "sre" - service: "vmalert-stateful-secured-supply-a-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: "false" - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-secured-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-supply-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml deleted file mode 100644 index 865ebcd..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/victoria-metrics-alert-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-supply-a-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-supply-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-supply-a-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-supply-a-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/supply/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-supply-a-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 1500m - memory: 3Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "supply" - team: "sre" - service: "vmalert-supply-a-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/gke-supply-prd-ase1a/victoriametrics-agent/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index 2e037b2..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,488 +0,0 @@ -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "supply-a-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -# -- Replica count -replicaCount: 2 - -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - -# -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables -deployment: - spec: - minReadySeconds: 1200 - progressDeadlineSeconds: 1800 - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - -image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - # -- Image tag, set to `Chart.AppVersion` by default - tag: v1.133.0 # rewrites Chart.AppVersion - # -- Variant of the image to use. - # e.g. enterprise, scratch - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - -# -- Image pull secrets -imagePullSecrets: [] - -# -- Add additional DNS entries to pods hosts file. Check [official documentation](https://kubernetes.io/docs/tasks/network/customize-hosts-file-for-pods/) -hostAliases: [] -# - ip: 192.168.1.1 -# hostNames: -# - test.example.com -# - another.example.net - -# -- Override chart name -nameOverride: "" - -# -- Override resources fullname -fullnameOverride: "vm-agent-supply-a-prd" - -# -- Container working directory -containerWorkingDir: "/" - -rbac: - # -- Enables Role/RoleBinding creation - create: true - - # -- Role/RoleBinding annotations - annotations: {} - - # -- Role/RoleBinding labels - extraLabels: {} - - # -- If true and `rbac.enabled`, will deploy a Role/RoleBinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - - # -- additional rules for a role - extraRules: [] - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - # -- Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-supl-sre-vmagnt-prd-mds@meesho-supply-prd-0622.iam.gserviceaccount.com - } - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # -- mount API token to pod directly - automountToken: true - -# -- See `kubectl explain poddisruptionbudget.spec` for more or check [official documentation](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# -- Generates `remoteWrite.*` flags and config maps with value content for values, that are of type list of map. -# Each item should contain `url` param to pass validation. -remoteWrite: - - url: http://vm-insert-supply-a-prd.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write - disableOnDiskQueue: false - dropSamplesOnOverload: false - # - url: http://vm-insert-supply-prd.victoriametrics.svc.clusterset.local:8480/insert/multitenant/prometheus/api/v1/write - # disableOnDiskQueue: false - # dropSamplesOnOverload: false - - url: http://prd-census-server-supply.prd-census-server-supply.svc.cluster.local/api/v1/write - disableOnDiskQueue: true - dropSamplesOnOverload: true -# urlRelabelConfig: -# - action: keep -# source_labels: [env] -# regex: "dev" - -# -- VMAgent extra command line arguments -extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8429 - promscrape.config.strictParse: false # default is true - promscrape.maxScrapeSize: 1000000000 # default is 167772160 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - # promscrape.suppressScrapeErrors: true # default is false - - # Uncomment and specify the port if you want to support any of the protocols: - # https://docs.victoriametrics.com/victoriametrics/vmagent/#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for more details. -env: - - name: GOGC - value: "200" - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth-secret - # key: password - -# -- Specify alternative source for env variables -envFrom: - [] - #- configMapRef: - # name: special-config - -# -- Extra labels for Deployment and Statefulset -extraLabels: - bu: "supply" - team: "supply-sre" - service: "vm-agent-supply-a-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra labels for Pods only -podLabels: - bu: "supply" - team: "supply-sre" - service: "vm-agent-supply-a-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra selector labels common for pod and service -selectorLabels: {} - -# -- Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# -- Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# -- Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -# -- Extra containers to run in a pod with vmagent -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -# -- Init containers for vmagent -initContainers: - [] - # - name: example - # image: example-image - -# -- Security context to be added to pod -podSecurityContext: - enabled: true - # fsGroup: 2000 - -# -- Security context to be added to pod's containers -securityContext: - enabled: true - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - # -- Enable agent service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - extraLabels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) for details - externalIPs: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8429 - # -- Target port - targetPort: http - # nodePort: 30000 - # -- Service type - type: ClusterIP - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - -ingress: - # -- Enable deployment of ingress for agent - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-agent-supply-a-prd.meeshogcp.in - path: - - / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - - # -- Ingress controller class name - ingressClassName: "nginx-internal" - - # -- Ingress path type - pathType: Prefix - -# -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) -resources: - requests: - cpu: 70 - memory: 100Gi - -# -- Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) -nodeSelector: - cloud.google.com/compute-class: "vmagent-n4" - -# -- Node tolerations for server scheduling to nodes with taints. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmagent-n4" - effect: "NoSchedule" - -# -- Pod topologySpreadConstraints -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# -- Pod affinity -affinity: {} - -# -- VMAgent [scraping configuration](https://docs.victoriametrics.com/victoriametrics/vmagent/#how-to-collect-metrics-in-prometheus-format) -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-supply-a-prd" - -# -- Priority class to be assigned to the pod(s) -priorityClassName: "" - -# -- Enable the host network -hostNetwork: false - -serviceMonitor: - # -- Enable deployment of Service Monitor for server component. This is Prometheus operator object - enabled: false - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Service Monitor relabelings - relabelings: [] - # -- Basic auth params for Service Monitor - basicAuth: {} - # -- Service Monitor metricRelabelings - metricRelabelings: [] - # -- Service Monitor targetPort - targetPort: http - # interval: 15s - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - -# -- Empty dir configuration for a case, when persistence is disabled -emptyDir: {} - -persistentVolume: - # -- Create/use Persistent Volume Claim for server component. Empty dir if false - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Size of the volume. Should be calculated based on the logs you send and retention policy you set. - size: 10Gi - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume additional labels - extraLabels: {} - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Horizontal Pod Autoscaling. -# Note that it is not intended to be used for vmagents which perform scraping. -# In order to scale scraping vmagents check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) -horizontalPodAutoscaling: - # -- Use HPA for vmagent - enabled: false - # -- Maximum replicas for HPA to use to to scale vmagent - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale vmagent - minReplicas: 1 - # -- Metric for HPA to use to scale vmagent - metrics: [] - -# -- VMAgent scrape configuration -config: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -probe: - # -- Readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - # -- Liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - # -- Startup probe - startup: {} - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -allowedMetricsEndpoints: - - /metrics - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-supply-prd-ase1a/victoriametrics-insert/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/victoriametrics-insert/custom-values.yaml deleted file mode 100644 index 177c4a7..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/victoriametrics-insert/custom-values.yaml +++ /dev/null @@ -1,411 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: false - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- IDs of vmstorage nodes to exclude from writing - excludeStorageIDs: [] - # -- Override default `app` label name - name: vminsert - # -- VMInsert strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - # override Chart.AppVersion - tag: v1.107.0-cluster - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMInsert http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-supply-a-prd - # -- Extra command line arguments for vminsert component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8480 - maxLabelsPerTimeseries: 65 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-supply-a-prd-0.vm-storage-supply-a-prd.victoriametrics.svc:8400" - - "vm-storage-supply-a-prd-1.vm-storage-supply-a-prd.victoriametrics.svc:8400" - - "vm-storage-supply-a-prd-2.vm-storage-supply-a-prd.victoriametrics.svc:8400" - - "vm-storage-supply-a-prd-3.vm-storage-supply-a-prd.victoriametrics.svc:8400" - - "vm-storage-supply-a-prd-4.vm-storage-supply-a-prd.victoriametrics.svc:8400" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "supply" - team: "supply-sre" - service: "vm-insert-supply-a-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Pod's additional labels - podLabels: - bu: "supply" - team: "supply-sre" - service: "vm-insert-supply-a-prd" - env: "prd" - priority: "p0" - type: "vminsert" - terminationGracePeriodSeconds: 30 - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - - # -- Readiness & Liveness probes - probe: - # -- VMInsert readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert startup probe - startup: {} - - # -- Relabel configuration - relabel: - enabled: false - config: [] - # -- Use existing configmap if specified - # otherwise .config values will be used. Relabel config **should** reside under `relabel.yml` key - configMap: "" - - # Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 8 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vminsert - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vminsert - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - cloud.google.com/compute-class: "vminsert-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 8 - memory: 8Gi - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - service: - # -- Create VMInsert service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here]( https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) for details - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-insert-supply-a-prd.meeshogcp.in - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - - # -- Ingress controller class name - ingressClassName: nginx-internal - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for insert component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/insert" .Values.vminsert }}' - - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vminsert component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/gke-supply-prd-ase1a/victoriametrics-select/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/victoriametrics-select/custom-values.yaml deleted file mode 100644 index cc19f0b..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/victoriametrics-select/custom-values.yaml +++ /dev/null @@ -1,449 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - - splitService: true - externalService: - enabled: true - name: vmselect-supply-a-prd-proxy - - # -- Enable clusternative service for LoadBalancer vmselect component - clusternativeService: - enabled: false - - # -- Override default `app` label name - name: "" - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # override Chart.AppVersion - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMSelect http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-supply-a-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - # -- Extra command line arguments for vmselect component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8481 - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-supply-a-prd-0.vm-storage-supply-a-prd.victoriametrics.svc:8401" - - "vm-storage-supply-a-prd-1.vm-storage-supply-a-prd.victoriametrics.svc:8401" - - "vm-storage-supply-a-prd-2.vm-storage-supply-a-prd.victoriametrics.svc:8401" - - "vm-storage-supply-a-prd-3.vm-storage-supply-a-prd.victoriametrics.svc:8401" - - "vm-storage-supply-a-prd-4.vm-storage-supply-a-prd.victoriametrics.svc:8401" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "supply" - team: "supply-sre" - service: "vm-select-supply-a-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # -- Pod's additional labels - podLabels: - bu: "supply" - team: "supply-sre" - service: "vm-select-supply-a-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Readiness & Liveness probes - probe: - # -- VMSelect readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 40 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 15 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vmselect - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vmselect - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - cloud.google.com/compute-class: "vmselect-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 18 - memory: 32Gi - - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Create VMSelect service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: false - - # -- Ingress annotations - annotations: {} - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-select-supply.prd.meesho.int - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # -- Ingress controller class name - ingressClassName: contour-internal-0 - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for select component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/select" .Values.vmselect }}' - - # -- vmselect mode: deployment, daemonSet - mode: deployment - - # -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables - deployment: - spec: - # -- VMSelect strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - - # -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables - statefulSet: - enabled: false - spec: - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume extra labels - extraLabels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmselect component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: false - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - diff --git a/helm-overrides/gke-supply-prd-ase1a/victoriametrics-storage/custom-values.yaml b/helm-overrides/gke-supply-prd-ase1a/victoriametrics-storage/custom-values.yaml deleted file mode 100644 index 3f603a4..0000000 --- a/helm-overrides/gke-supply-prd-ase1a/victoriametrics-storage/custom-values.yaml +++ /dev/null @@ -1,363 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - enabled: false - -vminsert: - enabled: false - -vmauth: - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- Override default `app` label name - name: vmstorage - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Variant of the image to use. e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMStorage http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-supply-a-prd - - # -- Additional environment variables (ex.: secret tokens, flags). Check https://docs.victoriametrics.com/victoriametrics/#environment-variables for details - env: [] - # -- Specify alternative source for env variables - envFrom: [] - # -- Data retention period. Possible units character: h(ours), d(ays), w(eeks), y(ears), if no unit character specified - month. The minimum retention period is 24h. - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - httpListenAddr: :8482 - loggerTimezone: "Asia/Kolkata" - dedup.minScrapeInterval: 60s - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "vmstorage-c4d" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - cloud.google.com/compute-class: "vmstorage-c4d" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - # -- Use an alternate scheduler, e.g. "stork". Check https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ for details - # schedulerName: - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Override Persistent Volume Claim name - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are https://kubernetes.io/docs/concepts/storage/persistent-volumes/ - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume extra labels - extraLabels: {} - # -- Storage class name. Will be empty if not set - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume - size: 7400Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "supply" - team: "supply-sre" - service: "vm-storage-supply-a-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - # -- Pod's additional labels - podLabels: - bu: "supply" - team: "supply-sre" - service: "vm-storage-supply-a-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Details are https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 58 - memory: 475Gi - - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: - enabled: false - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - podSecurityContext: - enabled: false - service: - enabled: true - # -- Service annotations - annotations: {} - # -- Service ClusterIP - clusterIP: None - # -- Service type - type: ClusterIP - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmselect - vmselectPort: 8401 - # -- Extra service ports - extraPorts: [] - # -- Health check node port for a service. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilyPolicy: "" - # -- List of service IP families. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilies: [] - # -- Traffic Distribution. Check https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution - trafficDistribution: "" - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - minReadySeconds: 5 - # -- Readiness probes - probe: - # -- VMStorage readiness probe - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMStorage startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmstorage component - enabled: false - # -- Maximum replicas for HPA to use to to scale the vmstorage component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmstorage component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmstorage component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: - scaleDown: - selectPolicy: Disabled - - vmbackupmanager: - # -- Enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-admin-prd-ase1/argocd-admin-prd/custom-values.yaml b/helm-overrides/k8s-admin-prd-ase1/argocd-admin-prd/custom-values.yaml index 9d845f2..4c8c29b 100644 --- a/helm-overrides/k8s-admin-prd-ase1/argocd-admin-prd/custom-values.yaml +++ b/helm-overrides/k8s-admin-prd-ase1/argocd-admin-prd/custom-values.yaml @@ -2,238 +2,86 @@ argo-cd: global: image: tag: "v2.13.8" - additionalLabels: - bu: infra - team: devops - podLabels: - bu: infra - team: devops - nodeSelector: - dedicated: devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" + + # Single-node homelab VM (8GB RAM / 6 cores, see claude.md) — no dedicated + # devops node pool here, so the GKE nodeSelector/toleration pair from the + # fleet's admin cluster doesn't apply. Every component below is trimmed to + # a single replica with small resource requests to fit the ~700MB total + # budget claude.md tracks for ArgoCD. + + # SSO deferred per claude.md ("not yet implemented") — Dex stays off until + # that's picked back up. Revisit this file when it is. dex: - enabled: true + enabled: false + + controller: + replicas: 1 resources: + requests: + cpu: 200m + memory: 400Mi limits: cpu: 500m - memory: 1Gi + memory: 768Mi + + redis-ha: + enabled: false + redis: + resources: requests: + cpu: 50m + memory: 64Mi + limits: + memory: 128Mi + + repoServer: + replicas: 1 + resources: + requests: + cpu: 100m + memory: 256Mi + limits: cpu: 300m memory: 512Mi - metrics: - enabled: true - podAnnotations: - prometheus.io/scrape: true - prometheus.io/path: /metrics - prometheus.io/port: 5558 - controller: - replicas: 2 - enableStatefulSet: true - podAnnotations: - prometheus.io/scrape: true - prometheus.io/path: /metrics - prometheus.io/port: 8082 - resources: - limits: - cpu: "7" - memory: "12Gi" - requests: - cpu: "6" - memory: "8Gi" - env: - - name: ARGOCD_CONTROLLER_REPLICAS - value: '2' - redis-ha: - enabled: true - repoServer: - autoscaling: - enabled: true - maxReplicas: 15 - minReplicas: 3 - targetMemoryUtilizationPercentage: 70 - targetCPUUtilizationPercentage: 60 - metrics: - enabled: true - serviceMonitor: - enabled: false - interval: 60s - podAnnotations: - prometheus.io/scrape: true - prometheus.io/path: /metrics - prometheus.io/port: 8084 - resources: - limits: - cpu: 2500m - memory: 4Gi - requests: - cpu: 1500m - memory: 3Gi - env: - - name: ARGOCD_HELM_ALLOW_CONCURRENCY - value: 'true' + server: - ingress: - annotations.nginx.ingress.kubernetes.io/force-ssl-redirect: false - annotations.nginx.ingress.kubernetes.io/rewrite-target: / - annotations.nginx.ingress.kubernetes.io/ssl-redirect: false - enabled: true - hostname: "argocd-prd.meeshogcp.in" - ingressClassName: nginx-internal - replicas: 3 - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 15 - targetMemoryUtilizationPercentage: 60 - targetCPUUtilizationPercentage: 60 + replicas: 1 extraArgs: - --insecure - podAnnotations: - prometheus.io/scrape: true - prometheus.io/path: /metrics - prometheus.io/port: 8083 + ingress: + enabled: true + ingressClassName: contour + # LAN hostname, matching claude.md's documented access URL. Add the + # Tailscale nip.io variant (argocd.100.90.248.118.nip.io) the same + # way Gitea does it if remote access is needed later. + hostname: "argocd.192.168.1.7.nip.io" resources: - limits: - cpu: 1500m - memory: 4Gi requests: - cpu: 1 - memory: 2Gi + cpu: 50m + memory: 128Mi + limits: + cpu: 200m + memory: 256Mi + + # Not used by this repo's Applications (plain Application manifests + # rendered by generic-argo-apps-chart, not the ApplicationSet CRD) and + # notifications has no configured trigger/service — both off to save RAM. applicationSet: - replicaCount: 2 - resources: - limits: - cpu: 500m - memory: 1Gi - requests: - cpu: 300m - memory: 512Mi + enabled: false notifications: - resources: - limits: - cpu: 500m - memory: 1Gi - requests: - cpu: 300m - memory: 512Mi + enabled: false configs: cm: - resource.customizations: | - keda.sh/ScaledObject: - health.lua: | - local hs = {} - local healthy = false - local degraded = false - local suspended = false - - if obj.status ~= nil then - if obj.status.conditions ~= nil then - for i, condition in ipairs(obj.status.conditions) do - if condition.status == "False" and condition.type == "Ready" then - degraded = true - hs.message = condition.message - end - if condition.status == "True" and condition.type == "Ready" then - healthy = true - hs.message = condition.message - end - if condition.status == "True" and condition.type == "Paused" then - suspended = true - hs.message = condition.message - end - end - end - end - - if degraded == true then - hs.status = "Degraded" - return hs - elseif healthy == true then - hs.status = "Healthy" - if suspended == true then - hs.message = "ScaledObject is paused as part of normal operations." - else - hs.message = "ScaledObject is active." - end - return hs - end - - hs.status = "Progressing" - hs.message = "Creating ScaledObject or waiting for conditions." - return hs - url: https://argocd-prd.meeshogcp.in + url: "https://argocd.192.168.1.7.nip.io" timeout.reconciliation: 3m timeout.reconciliation.jitter: 60s - statusbadge.enabled: "true" - help.chatUrl: "https://meesho.slack.com/archives/C021QNS6JLV" - help.chatText: "Chat now!" - dex.config: | - logger: - level: error - format: json - connectors: - - type: github - id: github - name: GitHub - loadAllGroups: true - admin.enabled: "true" - config: - clientID: 7f82547c0e671780cda5 - clientSecret: $github-sso-secret:dex.github.clientSecret - orgs: - - name: Meesho - params: - controller.sharding.algorithm: round-robin - controller.status.processors: '40' - controller.operation.processors: '20' - controller.repo.server.timeout.seconds: '60' - rbac: - policy.csv: | - p, role:admins, *, *, */*, allow - ## Policy for Admin-NoDelete role - p, role:admin-nodelete, *, get, *, allow - p, role:admin-nodelete, *, create, *, allow - p, role:admin-nodelete, *, update, *, allow - p, role:admin-nodelete, *, update, devops/argocd-*, deny - p, role:admin-nodelete, applications, override, *, allow - p, role:admin-nodelete, applications, sync, *, allow - p, role:admin-nodelete, applications, action/*, *, allow - p, role:admin-nodelete, applications, delete/*/Pod/*/*, */*, allow - g, Meesho:devops-new, role:admin-nodelete - - p, role:superstore, applications, *, farmiso/*, allow - - p, role:backend, applications, get, */*, allow - p, role:backend, applications, update, */*, allow - p, role:backend, applications, sync, */*, allow - - g, Meesho:devops, role:admins - g, Meesho:superstore, role:superstore - g, Meesho:backend, role:backend - p, role:intern, *, get, *, allow - g, Meesho:devops-interns, role:intern - - ## Policy: sync access to aurva apps - p, role:aurva-sync, applications, get, devops/aurva-dataplane-*, allow - p, role:aurva-sync, applications, sync, devops/aurva-dataplane-*, allow - p, role:aurva-sync, applications, action/apps/*/restart, devops/aurva-dataplane-*, allow - g, keshav.pandya@meesho.com, role:aurva-sync - + # No custom RBAC policy: single-user homelab, the initial admin secret + # (kubectl -n argocd get secret argocd-initial-admin-secret) is enough. + # The fleet's role:admins / role:backend / GitHub-team policy.csv and + # real teammate emails from the source cluster are dropped here. repositories: - gcp-devops-admin: - url: https://github.com/Meesho/gcp-devops-admin - gcp-argocd-apps: - url: https://github.com/Meesho/gcp-argocd-apps - gcp-service-helmcharts: - url: https://github.com/Meesho/gcp-service-helmcharts - sre-argo-apps: - url: https://github.com/Meesho/sre-argo-apps devops-infra-helm-charts: - url: https://github.com/Meesho/devops-infra-helm-charts + url: http://gitea.192.168.1.7.nip.io/mukul/devops-infra-helm-charts.git devops-infra-argo-config: - url: https://github.com/Meesho/devops-infra-argo-config + url: http://gitea.192.168.1.7.nip.io/mukul/devops-infra-argo-config.git diff --git a/helm-overrides/k8s-aurva-prd-ase1/contour-internal/custom-values.yaml b/helm-overrides/k8s-aurva-prd-ase1/contour-internal/custom-values.yaml deleted file mode 100644 index 987b879..0000000 --- a/helm-overrides/k8s-aurva-prd-ase1/contour-internal/custom-values.yaml +++ /dev/null @@ -1,77 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: aurva - team: devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - image: - tag: 1.28.0-debian-11-r8 - enabled: true - podLabels: - bu: aurva - team: devops - env: prd - kind: deployment - logLevel: error - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 3500m - memory: 2Gi - service: - tcpLB: true - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int-aurva-prd"}}}' - networking.gke.io/internal-load-balancer-allow-global-access: "true" - ports: - http: 80 - https: 443 - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-aurva-prd-ase1/rancher/custom-values.yaml b/helm-overrides/k8s-aurva-prd-ase1/rancher/custom-values.yaml deleted file mode 100644 index ecef735..0000000 --- a/helm-overrides/k8s-aurva-prd-ase1/rancher/custom-values.yaml +++ /dev/null @@ -1,17 +0,0 @@ -rancher: - rancherImage: 'rancher/rancher' - rancherImageTag: latest - replicas: 2 - resources: - requests: - memory: 2G - cpu: 1 - limits: - memory: 2G - cpu: 4 - hostname: "rancher.aurva-prd.meeshogcp.in" - ingress: - ingressClassName: contour-internal - postDelete: - enabled: false -bootstrapPassword: welcome@123 diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/README.md b/helm-overrides/k8s-central-mqkafka-prd-ase1/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/backend-nginx-cluster/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/backend-nginx-cluster/custom-values.yaml deleted file mode 100644 index b5e2ccf..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/backend-nginx-cluster/custom-values.yaml +++ /dev/null @@ -1,241 +0,0 @@ -ingress-nginx: - fullname: nginx-backend - tcp: - "9625": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-kafka-external4-bootstrap:9094 - "9626": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-1:9094 - "9627": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-2:9094 - "9628": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-3:9094 - "9629": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-4:9094 - "9630": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-5:9094 - "9631": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-6:9094 - "9632": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-7:9094 - "9633": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-8:9094 - "9634": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-9:9094 - "9635": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-10:9094 - "9636": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-11:9094 - "9637": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-12:9094 - "9638": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-13:9094 - "9639": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-14:9094 - "9640": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-15:9094 - "9641": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-16:9094 - "9642": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-17:9094 - "9643": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-18:9094 - "9644": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-19:9094 - "9645": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-20:9094 - "9646": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-21:9094 - "9647": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-22:9094 - "9648": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-23:9094 - "9649": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-24:9094 - "9650": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-25:9094 - "9651": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-26:9094 - "9652": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-27:9094 - "9653": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-28:9094 - "9654": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-29:9094 - "9655": prd-backend-shared-cluster-1/prd-backend-shared-cluster-1-broker-external4-30:9094 - "9101": prd-backend-shared-cluster-2/backend-shared-cluster-2-kafka-external4-bootstrap:9094 - "9103": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-1:9094 - "9104": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-2:9094 - "9105": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-3:9094 - "9106": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-4:9094 - "9107": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-5:9094 - "9108": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-6:9094 - "9109": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-7:9094 - "9110": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-8:9094 - "9111": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-9:9094 - "9112": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-10:9094 - "9113": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-11:9094 - "9114": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-12:9094 - "9115": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-13:9094 - "9116": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-14:9094 - "9117": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-15:9094 - "9118": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-16:9094 - "9119": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-17:9094 - "9120": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-18:9094 - "9121": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-19:9094 - "9122": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-20:9094 - "9123": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-21:9094 - "9124": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-22:9094 - "9125": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-23:9094 - "9126": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-24:9094 - "9127": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-25:9094 - "9128": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-26:9094 - "9129": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-27:9094 - "9130": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-28:9094 - "9131": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-29:9094 - "9132": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-30:9094 - "9133": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-31:9094 - "9134": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-32:9094 - "9135": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-33:9094 - "9136": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-34:9094 - "9137": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-35:9094 - "9138": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-36:9094 - "9139": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-37:9094 - "9140": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-38:9094 - "9141": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-39:9094 - "9142": prd-backend-shared-cluster-2/backend-shared-cluster-2-broker-external4-40:9094 - "6322": prd-backend-shared-cluster-3/backend-shared-cluster-3-kafka-external4-bootstrap:9094 - "6323": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-0:9094 - "6324": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-1:9094 - "6325": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-2:9094 - "6326": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-3:9094 - "6327": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-4:9094 - "6328": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-5:9094 - "6329": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-6:9094 - "6330": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-7:9094 - "6331": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-8:9094 - "6332": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-9:9094 - "6333": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-10:9094 - "6334": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-11:9094 - "6335": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-12:9094 - "6336": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-13:9094 - "6337": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-14:9094 - "6338": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-15:9094 - "6339": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-16:9094 - "6340": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-17:9094 - "6341": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-18:9094 - "6342": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-19:9094 - "6343": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-20:9094 - "6344": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-21:9094 - "6345": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-22:9094 - "6346": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-23:9094 - "6347": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-24:9094 - "6348": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-25:9094 - "6349": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-26:9094 - "6350": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-27:9094 - "6351": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-28:9094 - "6352": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-29:9094 - "6353": prd-backend-shared-cluster-3/backend-shared-cluster-3-broker-external4-30:9094 - "6726": prd-backend-shared-cluster-4/backend-shared-cluster-4-kafka-external4-bootstrap:9094 - "6727": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-0:9094 - "6728": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-1:9094 - "6729": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-2:9094 - "6730": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-3:9094 - "6731": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-4:9094 - "6732": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-5:9094 - "6733": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-6:9094 - "6734": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-7:9094 - "6735": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-8:9094 - "6736": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-9:9094 - "6737": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-10:9094 - "6738": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-11:9094 - "6739": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-12:9094 - "6740": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-13:9094 - "6741": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-14:9094 - "6742": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-15:9094 - "6743": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-16:9094 - "6744": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-17:9094 - "6745": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-18:9094 - "6746": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-19:9094 - "6747": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-20:9094 - "6748": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-21:9094 - "6749": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-22:9094 - "6750": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-23:9094 - "6751": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-24:9094 - "6752": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-25:9094 - "6753": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-26:9094 - "6754": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-27:9094 - "6755": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-28:9094 - "6756": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-29:9094 - "6757": prd-backend-shared-cluster-4/backend-shared-cluster-4-broker-external4-30:9094 - "7029": prd-backend-low-latency-cluster/backend-low-latency-cluster-kafka-external4-bootstrap:9094 - "7030": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-0:9094 - "7031": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-1:9094 - "7032": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-2:9094 - "7033": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-3:9094 - "7034": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-4:9094 - "7035": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-5:9094 - "7036": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-6:9094 - "7037": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-7:9094 - "7038": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-8:9094 - "7039": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-9:9094 - "7040": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-10:9094 - "7041": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-11:9094 - "7042": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-12:9094 - "7043": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-13:9094 - "7044": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-14:9094 - "7045": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-15:9094 - "7046": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-16:9094 - "7047": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-17:9094 - "7048": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-18:9094 - "7049": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-19:9094 - "7050": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-20:9094 - "7051": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-21:9094 - "7052": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-22:9094 - "7053": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-23:9094 - "7054": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-24:9094 - "7055": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-25:9094 - "7056": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-26:9094 - "7057": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-27:9094 - "7058": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-28:9094 - "7059": prd-backend-low-latency-cluster/backend-low-latency-cluster-broker-external4-29:9094 - "7755": prd-backend-shared-cluster-5/backend-shared-cluster-5-kafka-external4-bootstrap:9094 - "7757": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-1:9094 - "7758": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-2:9094 - "7759": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-3:9094 - "7760": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-4:9094 - "7761": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-5:9094 - "7762": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-6:9094 - "7763": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-7:9094 - "7764": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-8:9094 - "7765": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-9:9094 - "7766": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-10:9094 - "7767": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-11:9094 - "7768": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-12:9094 - "7769": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-13:9094 - "7770": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-14:9094 - "7771": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-15:9094 - "7772": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-16:9094 - "7773": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-17:9094 - "7774": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-18:9094 - "7775": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-19:9094 - "7776": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-20:9094 - "7777": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-21:9094 - "7778": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-22:9094 - "7779": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-23:9094 - "7780": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-24:9094 - "7781": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-25:9094 - "7782": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-26:9094 - "7783": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-27:9094 - "7784": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-28:9094 - "7785": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-29:9094 - "7786": prd-backend-shared-cluster-5/backend-shared-cluster-5-broker-external4-30:9094 - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 5 - maxReplicas: 100 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - extraArgs: - enable-ssl-passthrough: true - ingressClassResource: - name: nginx-backend - service: - type: LoadBalancer - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-backend-tcp"}}}' - networking.gke.io/load-balancer-type: Internal - networking.gke.io/internal-load-balancer-allow-global-access: "true" - nodeSelector: - dedicated: nginx-backend - tolerations: - - key: "dedicated" - operator: "Equal" - value: "nginx-backend" - effect: "NoSchedule" - resources: - limits: - cpu: 2000m - memory: 2000Mi - requests: - cpu: 1000m - memory: 1500Mi - autoscaling: - enabled: true - minReplicas: 5 - maxReplicas: 100 diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/cert-manager/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/cert-manager/custom-values.yaml deleted file mode 100644 index 5196b61..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/cert-manager/custom-values.yaml +++ /dev/null @@ -1,129 +0,0 @@ -global: - logLevel: 2 - rbac: - create: true - priorityClassName: "high-priority" -installCRDs: false - -crds: - enabled: true - keep: true - -# Cert-manager Controller -replicaCount: 1 -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-controller - tag: v1.20.1 - pullPolicy: IfNotPresent - -nodeSelector: - dedicated: mqkafka-devops - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "mqkafka-devops" - effect: "NoSchedule" - -resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 500m - memory: 512Mi - -# Webhook Configuration -webhook: - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-webhook - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: contour-internal-0 - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "contour-internal-0" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# CA Injector Configuration -cainjector: - enabled: true - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-cainjector - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: contour-internal-0 - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "contour-internal-0" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# ACME Solver Configuration -acmesolver: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-acmesolver - tag: v1.20.1 - pullPolicy: IfNotPresent - -# Startup API Check -startupapicheck: - enabled: true - timeout: 1m - backoffLimit: 4 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-startupapicheck - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: contour-internal-0 - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "contour-internal-0" - effect: "NoSchedule" - - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - -# Prometheus Monitoring -prometheus: - enabled: true - servicemonitor: - enabled: false - interval: 60s - scrapeTimeout: 30s - labels: - prometheus: cert-manager diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/conntrack-adjuster/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 7b64b15..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - # Maximum number of conntrack entries - max: 2097152 - # Hash size for conntrack - hashsize: 524288 - # Sleep interval between adjustments (seconds) - sleepInterval: 30 - -# Enable additional matchExpressions -# addExtraMatchExpressions: true - -# Additional matchExpressions to append -# additionalMatchExpressions: -# - key: node-role -# operator: In -# values: -# - "worker" -# - "ingress" -# - key: environment -# operator: In -# values: -# - "production" -# - "staging" \ No newline at end of file diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/contour-internal-0/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/contour-internal-0/custom-values.yaml deleted file mode 100644 index c595c4e..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,97 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: central-mq - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 4Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - updateStrategy: - rollingUpdate: - maxSurge: 10% - enabled: true - podLabels: - bu: central-mq - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 350 - targetCPU: "55" - targetMemory: "55" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 12 - memory: 8Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int-0-central-mq-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/ds-kafka-clusters-nginx/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/ds-kafka-clusters-nginx/custom-values.yaml deleted file mode 100644 index 0681f2b..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/ds-kafka-clusters-nginx/custom-values.yaml +++ /dev/null @@ -1,75 +0,0 @@ -ingress-nginx: - fullname: nginx-ds - tcp: - "9525": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-kafka-external4-bootstrap:9094 - "9526": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-1:9094 - "9527": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-2:9094 - "9528": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-3:9094 - "9529": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-4:9094 - "9530": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-5:9094 - "9531": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-6:9094 - "9532": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-7:9094 - "9533": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-8:9094 - "9534": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-9:9094 - "9535": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-10:9094 - "9536": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-11:9094 - "9537": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-12:9094 - "9538": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-13:9094 - "9539": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-14:9094 - "9540": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-15:9094 - "9541": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-16:9094 - "9542": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-17:9094 - "9543": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-18:9094 - "9544": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-19:9094 - "9545": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-20:9094 - "9546": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-21:9094 - "9547": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-22:9094 - "9548": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-23:9094 - "9549": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-24:9094 - "9550": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-25:9094 - "9551": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-26:9094 - "9552": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-27:9094 - "9553": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-28:9094 - "9554": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-29:9094 - "9555": prd-ds-shared-cluster-1/prd-ds-shared-cluster-1-broker-external4-30:9094 - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 5 - maxReplicas: 15 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - extraArgs: - enable-ssl-passthrough: true - tcp-services-configmap: nginx-ds/tcp-services - ingressClassResource: - name: nginx-ds - service: - type: LoadBalancer - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-ds-tcp"}}}' - networking.gke.io/load-balancer-type: Internal - networking.gke.io/internal-load-balancer-allow-global-access: "true" - nodeSelector: - dedicated: nginx-ds - tolerations: - - key: "dedicated" - operator: "Equal" - value: "nginx-ds" - effect: "NoSchedule" - resources: - limits: - cpu: 2000m - memory: 2000Mi - requests: - cpu: 1000m - memory: 1500Mi - autoscaling: - enabled: true - minReplicas: 5 - maxReplicas: 100 diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/external-secrets/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/external-secrets/custom-values.yaml deleted file mode 100644 index 62dbf34..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/external-secrets/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Additional service annotations - annotations: {} - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: mq-devops - nodeSelector: - dedicated: mq-devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: mqkafka-devops - nodeSelector: - dedicated: mqkafka-devops - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: mqkafka-devops - nodeSelector: - dedicated: mqkafka-devops diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/flagger/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/flagger/custom-values.yaml deleted file mode 100644 index 0b9e3a3..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/flagger/custom-values.yaml +++ /dev/null @@ -1,55 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: hpa-changes-26 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - dedicated: mqkafka-devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: mqkafka-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: devops - bu: central diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/fluentd/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/fluentd/custom-values.yaml deleted file mode 100644 index e7aa126..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/fluentd/custom-values.yaml +++ /dev/null @@ -1,720 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-dpspsre-fluentd-prd@meesho-mqkafka-prd-0225.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: central - team: sre - type: fluentd - service: fluentd-mqkafka-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: central - team: sre - type: fluentd - service: fluentd-mqkafka-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - - -# externalSecret: -# enabled: false -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: [] -# - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/ingress-nginx-internal-v1/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/ingress-nginx-internal-v1/custom-values.yaml deleted file mode 100644 index 1ae4919..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/ingress-nginx-internal-v1/custom-values.yaml +++ /dev/null @@ -1,43 +0,0 @@ -ingress-nginx: - fullname: nginx-controller-v1 - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 5 - maxReplicas: 15 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - extraArgs: - enable-ssl-passthrough: true - tcp-services-configmap: nginx-prd/tcp-services - ingressClassResource: - name: nginx-internal-v1 - service: - type: LoadBalancer - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-ctrl-v1-tcp"}}}' - networking.gke.io/load-balancer-type: Internal - networking.gke.io/internal-load-balancer-allow-global-access: "true" - nodeSelector: - dedicated: mq-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "mq-devops" - effect: "NoSchedule" - resources: - limits: - cpu: 2000m - memory: 2000Mi - requests: - cpu: 1000m - memory: 1500Mi - autoscaling: - enabled: true - minReplicas: 5 - maxReplicas: 100 diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/ingress-nginx-internal/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/ingress-nginx-internal/custom-values.yaml deleted file mode 100644 index 9525b5c..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/ingress-nginx-internal/custom-values.yaml +++ /dev/null @@ -1,43 +0,0 @@ -ingress-nginx: - fullname: nginx-int-mq-controller - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 5 - maxReplicas: 15 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - extraArgs: - enable-ssl-passthrough: true - tcp-services-configmap: nginx-int/tcp-services - ingressClassResource: - name: nginx-internal - service: - type: LoadBalancer - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-internal-mqkafka-prd-tcp"}}}' - networking.gke.io/load-balancer-type: Internal - networking.gke.io/internal-load-balancer-allow-global-access: "true" - nodeSelector: - dedicated: mq-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "mq-devops" - effect: "NoSchedule" - resources: - limits: - cpu: 2000m - memory: 2000Mi - requests: - cpu: 1000m - memory: 1500Mi - autoscaling: - enabled: true - minReplicas: 12 - maxReplicas: 100 diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/keda/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/keda/custom-values.yaml deleted file mode 100644 index 281a6bc..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/keda/custom-values.yaml +++ /dev/null @@ -1,38 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: mqkafka-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "mqkafka-devops" - effect: "NoSchedule" - podLabels: - bu: "central" - team: "mqkafka-devops" - metricsAdapter: - bu: "central" - team: "mqkafka-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 2000Mi - requests: - cpu: 250m - memory: 300Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 30m - memory: 100Mi - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' \ No newline at end of file diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/kube-events/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/kube-events/custom-values.yaml deleted file mode 100644 index 574e9fc..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-mqkafka-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - dedicated: mqkafka-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: mqkafka-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: mqkafka-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: mqkafka-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: mqkafka-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: mqkafka-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 0d900ae..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-mq-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-mq-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "central" - team: "mq" - service: "kube-state-metrics-mq-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "victoriametrics" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 600m - memory: 1200Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/nginx-demand/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/nginx-demand/custom-values.yaml deleted file mode 100644 index 36f1c86..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/nginx-demand/custom-values.yaml +++ /dev/null @@ -1,228 +0,0 @@ -ingress-nginx: - fullname: nginx-demand - tcp: - "6201": "prd-comms-backend-cluster-1/comms-backend-cluster-1-kafka-external4-bootstrap:9094" - "6204": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-1:9094" - "6205": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-2:9094" - "6206": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-3:9094" - "6207": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-4:9094" - "6208": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-5:9094" - "6209": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-6:9094" - "6210": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-7:9094" - "6211": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-8:9094" - "6212": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-9:9094" - "6213": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-10:9094" - "6214": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-11:9094" - "6215": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-12:9094" - "6216": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-13:9094" - "6217": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-14:9094" - "6218": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-15:9094" - "6219": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-16:9094" - "6220": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-17:9094" - "6221": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-18:9094" - "6222": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-19:9094" - "6223": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-20:9094" - "6224": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-21:9094" - "6225": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-22:9094" - "6226": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-23:9094" - "6227": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-24:9094" - "6228": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-25:9094" - "6229": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-26:9094" - "6230": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-27:9094" - "6231": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-28:9094" - "6232": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-29:9094" - "6233": "prd-comms-backend-cluster-1/comms-backend-cluster-1-broker-external4-30:9094" - "6825": "prd-comms-backend-cluster-2/comms-backend-cluster-2-kafka-external4-bootstrap:9094" - "6829": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-1:9094" - "6830": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-2:9094" - "6831": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-3:9094" - "6832": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-4:9094" - "6833": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-5:9094" - "6834": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-6:9094" - "6835": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-7:9094" - "6836": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-8:9094" - "6837": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-9:9094" - "6838": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-10:9094" - "6839": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-11:9094" - "6840": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-12:9094" - "6841": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-13:9094" - "6842": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-14:9094" - "6843": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-15:9094" - "6844": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-16:9094" - "6845": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-17:9094" - "6846": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-18:9094" - "6847": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-19:9094" - "6848": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-20:9094" - "6849": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-21:9094" - "6850": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-22:9094" - "6851": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-23:9094" - "6852": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-24:9094" - "6853": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-25:9094" - "6854": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-26:9094" - "6855": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-27:9094" - "6856": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-28:9094" - "6857": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-29:9094" - "6858": "prd-comms-backend-cluster-2/comms-backend-cluster-2-broker-external4-30:9094" - "6421": "prd-backend-discovery-1/backend-discovery-1-kafka-external4-bootstrap:9094" - "6425": "prd-backend-discovery-1/backend-discovery-1-broker-external4-1:9094" - "6426": "prd-backend-discovery-1/backend-discovery-1-broker-external4-2:9094" - "6427": "prd-backend-discovery-1/backend-discovery-1-broker-external4-3:9094" - "6428": "prd-backend-discovery-1/backend-discovery-1-broker-external4-4:9094" - "6429": "prd-backend-discovery-1/backend-discovery-1-broker-external4-5:9094" - "6430": "prd-backend-discovery-1/backend-discovery-1-broker-external4-6:9094" - "6431": "prd-backend-discovery-1/backend-discovery-1-broker-external4-7:9094" - "6432": "prd-backend-discovery-1/backend-discovery-1-broker-external4-8:9094" - "6433": "prd-backend-discovery-1/backend-discovery-1-broker-external4-9:9094" - "6434": "prd-backend-discovery-1/backend-discovery-1-broker-external4-10:9094" - "6435": "prd-backend-discovery-1/backend-discovery-1-broker-external4-11:9094" - "6436": "prd-backend-discovery-1/backend-discovery-1-broker-external4-12:9094" - "6437": "prd-backend-discovery-1/backend-discovery-1-broker-external4-13:9094" - "6438": "prd-backend-discovery-1/backend-discovery-1-broker-external4-14:9094" - "6439": "prd-backend-discovery-1/backend-discovery-1-broker-external4-15:9094" - "6440": "prd-backend-discovery-1/backend-discovery-1-broker-external4-16:9094" - "6441": "prd-backend-discovery-1/backend-discovery-1-broker-external4-17:9094" - "6442": "prd-backend-discovery-1/backend-discovery-1-broker-external4-18:9094" - "6443": "prd-backend-discovery-1/backend-discovery-1-broker-external4-19:9094" - "6444": "prd-backend-discovery-1/backend-discovery-1-broker-external4-20:9094" - "6445": "prd-backend-discovery-1/backend-discovery-1-broker-external4-21:9094" - "6446": "prd-backend-discovery-1/backend-discovery-1-broker-external4-22:9094" - "6447": "prd-backend-discovery-1/backend-discovery-1-broker-external4-23:9094" - "6448": "prd-backend-discovery-1/backend-discovery-1-broker-external4-24:9094" - "6449": "prd-backend-discovery-1/backend-discovery-1-broker-external4-25:9094" - "6450": "prd-backend-discovery-1/backend-discovery-1-broker-external4-26:9094" - "6451": "prd-backend-discovery-1/backend-discovery-1-broker-external4-27:9094" - "6452": "prd-backend-discovery-1/backend-discovery-1-broker-external4-28:9094" - "6453": "prd-backend-discovery-1/backend-discovery-1-broker-external4-29:9094" - "6454": "prd-backend-discovery-1/backend-discovery-1-broker-external4-30:9094" - "6625": "prd-backend-discovery-2/backend-discovery-2-kafka-external4-bootstrap:9094" - "6627": "prd-backend-discovery-2/backend-discovery-2-broker-external4-1:9094" - "6628": "prd-backend-discovery-2/backend-discovery-2-broker-external4-2:9094" - "6629": "prd-backend-discovery-2/backend-discovery-2-broker-external4-3:9094" - "6630": "prd-backend-discovery-2/backend-discovery-2-broker-external4-4:9094" - "6631": "prd-backend-discovery-2/backend-discovery-2-broker-external4-5:9094" - "6632": "prd-backend-discovery-2/backend-discovery-2-broker-external4-6:9094" - "6633": "prd-backend-discovery-2/backend-discovery-2-broker-external4-7:9094" - "6634": "prd-backend-discovery-2/backend-discovery-2-broker-external4-8:9094" - "6635": "prd-backend-discovery-2/backend-discovery-2-broker-external4-9:9094" - "6636": "prd-backend-discovery-2/backend-discovery-2-broker-external4-10:9094" - "6637": "prd-backend-discovery-2/backend-discovery-2-broker-external4-11:9094" - "6638": "prd-backend-discovery-2/backend-discovery-2-broker-external4-12:9094" - "6639": "prd-backend-discovery-2/backend-discovery-2-broker-external4-13:9094" - "6640": "prd-backend-discovery-2/backend-discovery-2-broker-external4-14:9094" - "6641": "prd-backend-discovery-2/backend-discovery-2-broker-external4-15:9094" - "6642": "prd-backend-discovery-2/backend-discovery-2-broker-external4-16:9094" - "6643": "prd-backend-discovery-2/backend-discovery-2-broker-external4-17:9094" - "6644": "prd-backend-discovery-2/backend-discovery-2-broker-external4-18:9094" - "6645": "prd-backend-discovery-2/backend-discovery-2-broker-external4-19:9094" - "6646": "prd-backend-discovery-2/backend-discovery-2-broker-external4-20:9094" - "6647": "prd-backend-discovery-2/backend-discovery-2-broker-external4-21:9094" - "6648": "prd-backend-discovery-2/backend-discovery-2-broker-external4-22:9094" - "6649": "prd-backend-discovery-2/backend-discovery-2-broker-external4-23:9094" - "6650": "prd-backend-discovery-2/backend-discovery-2-broker-external4-24:9094" - "6651": "prd-backend-discovery-2/backend-discovery-2-broker-external4-25:9094" - "6652": "prd-backend-discovery-2/backend-discovery-2-broker-external4-26:9094" - "6653": "prd-backend-discovery-2/backend-discovery-2-broker-external4-27:9094" - "6654": "prd-backend-discovery-2/backend-discovery-2-broker-external4-28:9094" - "6655": "prd-backend-discovery-2/backend-discovery-2-broker-external4-29:9094" - "6656": "prd-backend-discovery-2/backend-discovery-2-broker-external4-30:9094" - "6928": "prd-backend-discovery-3/backend-discovery-3-kafka-external4-bootstrap:9094" - "6930": "prd-backend-discovery-3/backend-discovery-3-broker-external4-1:9094" - "6931": "prd-backend-discovery-3/backend-discovery-3-broker-external4-2:9094" - "6932": "prd-backend-discovery-3/backend-discovery-3-broker-external4-3:9094" - "6933": "prd-backend-discovery-3/backend-discovery-3-broker-external4-4:9094" - "6934": "prd-backend-discovery-3/backend-discovery-3-broker-external4-5:9094" - "6935": "prd-backend-discovery-3/backend-discovery-3-broker-external4-6:9094" - "6936": "prd-backend-discovery-3/backend-discovery-3-broker-external4-7:9094" - "6937": "prd-backend-discovery-3/backend-discovery-3-broker-external4-8:9094" - "6938": "prd-backend-discovery-3/backend-discovery-3-broker-external4-9:9094" - "6939": "prd-backend-discovery-3/backend-discovery-3-broker-external4-10:9094" - "6940": "prd-backend-discovery-3/backend-discovery-3-broker-external4-11:9094" - "6941": "prd-backend-discovery-3/backend-discovery-3-broker-external4-12:9094" - "6942": "prd-backend-discovery-3/backend-discovery-3-broker-external4-13:9094" - "6943": "prd-backend-discovery-3/backend-discovery-3-broker-external4-14:9094" - "6944": "prd-backend-discovery-3/backend-discovery-3-broker-external4-15:9094" - "6945": "prd-backend-discovery-3/backend-discovery-3-broker-external4-16:9094" - "6946": "prd-backend-discovery-3/backend-discovery-3-broker-external4-17:9094" - "6947": "prd-backend-discovery-3/backend-discovery-3-broker-external4-18:9094" - "6948": "prd-backend-discovery-3/backend-discovery-3-broker-external4-19:9094" - "6949": "prd-backend-discovery-3/backend-discovery-3-broker-external4-20:9094" - "6950": "prd-backend-discovery-3/backend-discovery-3-broker-external4-21:9094" - "6951": "prd-backend-discovery-3/backend-discovery-3-broker-external4-22:9094" - "6952": "prd-backend-discovery-3/backend-discovery-3-broker-external4-23:9094" - "6953": "prd-backend-discovery-3/backend-discovery-3-broker-external4-24:9094" - "6954": "prd-backend-discovery-3/backend-discovery-3-broker-external4-25:9094" - "6955": "prd-backend-discovery-3/backend-discovery-3-broker-external4-26:9094" - "6956": "prd-backend-discovery-3/backend-discovery-3-broker-external4-27:9094" - "6957": "prd-backend-discovery-3/backend-discovery-3-broker-external4-28:9094" - "6958": "prd-backend-discovery-3/backend-discovery-3-broker-external4-29:9094" - "6959": "prd-backend-discovery-3/backend-discovery-3-broker-external4-30:9094" - "7553": "prd-backend-real-time-indexing/backend-real-time-indexing-kafka-external4-bootstrap:9094" - "7555": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-1:9094" - "7556": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-2:9094" - "7557": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-3:9094" - "7558": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-4:9094" - "7559": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-5:9094" - "7560": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-6:9094" - "7561": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-7:9094" - "7562": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-8:9094" - "7563": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-9:9094" - "7564": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-10:9094" - "7565": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-11:9094" - "7566": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-12:9094" - "7567": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-13:9094" - "7568": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-14:9094" - "7569": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-15:9094" - "7570": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-16:9094" - "7571": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-17:9094" - "7572": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-18:9094" - "7573": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-19:9094" - "7574": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-20:9094" - "7575": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-21:9094" - "7576": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-22:9094" - "7577": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-23:9094" - "7578": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-24:9094" - "7579": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-25:9094" - "7580": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-26:9094" - "7581": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-27:9094" - "7582": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-28:9094" - "7583": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-29:9094" - "7584": "prd-backend-real-time-indexing/backend-real-time-indexing-broker-external4-30:9094" - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 5 - maxReplicas: 100 - targetCPUUtilizationPercentage: 40 - extraArgs: - enable-ssl-passthrough: true - ingressClassResource: - name: nginx-demand - service: - type: LoadBalancer - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-demand-tcp"}}}' - networking.gke.io/load-balancer-type: Internal - networking.gke.io/internal-load-balancer-allow-global-access: "true" - nodeSelector: - dedicated: nginx-demand - tolerations: - - key: "dedicated" - operator: "Equal" - value: "nginx-demand" - effect: "NoSchedule" - resources: - limits: - cpu: 2000m - memory: 2000Mi - requests: - cpu: 1000m - memory: 1500Mi - autoscaling: - enabled: true - minReplicas: 5 - maxReplicas: 100 diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/nginx-supply/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/nginx-supply/custom-values.yaml deleted file mode 100644 index d7856ab..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/nginx-supply/custom-values.yaml +++ /dev/null @@ -1,206 +0,0 @@ -ingress-nginx: - fullname: nginx-supply - tcp: - "7351": "prd-backend-ads-1/backend-ads-1-kafka-external4-bootstrap:9094" - "7353": "prd-backend-ads-1/backend-ads-1-broker-external4-1:9094" - "7354": "prd-backend-ads-1/backend-ads-1-broker-external4-2:9094" - "7355": "prd-backend-ads-1/backend-ads-1-broker-external4-3:9094" - "7356": "prd-backend-ads-1/backend-ads-1-broker-external4-4:9094" - "7357": "prd-backend-ads-1/backend-ads-1-broker-external4-5:9094" - "7358": "prd-backend-ads-1/backend-ads-1-broker-external4-6:9094" - "7359": "prd-backend-ads-1/backend-ads-1-broker-external4-7:9094" - "7360": "prd-backend-ads-1/backend-ads-1-broker-external4-8:9094" - "7361": "prd-backend-ads-1/backend-ads-1-broker-external4-9:9094" - "7362": "prd-backend-ads-1/backend-ads-1-broker-external4-10:9094" - "7363": "prd-backend-ads-1/backend-ads-1-broker-external4-11:9094" - "7364": "prd-backend-ads-1/backend-ads-1-broker-external4-12:9094" - "7365": "prd-backend-ads-1/backend-ads-1-broker-external4-13:9094" - "7366": "prd-backend-ads-1/backend-ads-1-broker-external4-14:9094" - "7367": "prd-backend-ads-1/backend-ads-1-broker-external4-15:9094" - "7368": "prd-backend-ads-1/backend-ads-1-broker-external4-16:9094" - "7369": "prd-backend-ads-1/backend-ads-1-broker-external4-17:9094" - "7370": "prd-backend-ads-1/backend-ads-1-broker-external4-18:9094" - "7371": "prd-backend-ads-1/backend-ads-1-broker-external4-19:9094" - "7372": "prd-backend-ads-1/backend-ads-1-broker-external4-20:9094" - "7373": "prd-backend-ads-1/backend-ads-1-broker-external4-21:9094" - "7374": "prd-backend-ads-1/backend-ads-1-broker-external4-22:9094" - "7375": "prd-backend-ads-1/backend-ads-1-broker-external4-23:9094" - "7376": "prd-backend-ads-1/backend-ads-1-broker-external4-24:9094" - "7377": "prd-backend-ads-1/backend-ads-1-broker-external4-25:9094" - "7378": "prd-backend-ads-1/backend-ads-1-broker-external4-26:9094" - "7379": "prd-backend-ads-1/backend-ads-1-broker-external4-27:9094" - "7380": "prd-backend-ads-1/backend-ads-1-broker-external4-28:9094" - "7381": "prd-backend-ads-1/backend-ads-1-broker-external4-29:9094" - "7382": "prd-backend-ads-1/backend-ads-1-broker-external4-30:9094" - "7231": "prd-supply-backend-shared-1/supply-backend-shared-1-kafka-external4-bootstrap:9094" - "7233": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-1:9094" - "7234": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-2:9094" - "7235": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-3:9094" - "7236": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-4:9094" - "7237": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-5:9094" - "7238": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-6:9094" - "7239": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-7:9094" - "7240": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-8:9094" - "7241": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-9:9094" - "7242": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-10:9094" - "7243": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-11:9094" - "7244": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-12:9094" - "7245": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-13:9094" - "7246": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-14:9094" - "7247": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-15:9094" - "7248": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-16:9094" - "7249": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-17:9094" - "7250": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-18:9094" - "7251": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-19:9094" - "7252": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-20:9094" - "7253": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-21:9094" - "7254": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-22:9094" - "7255": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-23:9094" - "7256": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-24:9094" - "7257": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-25:9094" - "7258": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-26:9094" - "7259": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-27:9094" - "7260": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-28:9094" - "7261": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-29:9094" - "7262": "prd-supply-backend-shared-1/supply-backend-shared-1-broker-external4-30:9094" - "7654": "prd-backend-fulfillment-1/backend-fulfillment-1-kafka-external4-bootstrap:9094" - "7656": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-1:9094" - "7657": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-2:9094" - "7658": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-3:9094" - "7659": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-4:9094" - "7660": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-5:9094" - "7661": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-6:9094" - "7662": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-7:9094" - "7663": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-8:9094" - "7664": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-9:9094" - "7665": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-10:9094" - "7666": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-11:9094" - "7667": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-12:9094" - "7668": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-13:9094" - "7669": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-14:9094" - "7670": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-15:9094" - "7671": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-16:9094" - "7672": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-17:9094" - "7673": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-18:9094" - "7674": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-19:9094" - "7675": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-20:9094" - "7676": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-21:9094" - "7677": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-22:9094" - "7678": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-23:9094" - "7679": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-24:9094" - "7680": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-25:9094" - "7681": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-26:9094" - "7682": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-27:9094" - "7683": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-28:9094" - "7684": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-29:9094" - "7685": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-30:9094" - "7686": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-31:9094" - "7687": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-32:9094" - "7688": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-33:9094" - "7689": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-34:9094" - "7690": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-35:9094" - "7691": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-36:9094" - "7692": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-37:9094" - "7693": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-38:9094" - "7694": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-39:9094" - "7695": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-40:9094" - "7696": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-41:9094" - "7697": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-42:9094" - "7698": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-43:9094" - "7699": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-44:9094" - "7700": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-45:9094" - "7701": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-46:9094" - "7702": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-47:9094" - "7703": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-48:9094" - "7704": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-49:9094" - "7705": "prd-backend-fulfillment-1/backend-fulfillment-1-broker-external4-50:9094" - "8501": "prd-backend-fulfillment-2/backend-fulfillment-2-kafka-external4-bootstrap:9094" - "8503": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-1:9094" - "8504": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-2:9094" - "8505": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-3:9094" - "8506": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-4:9094" - "8507": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-5:9094" - "8508": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-6:9094" - "8509": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-7:9094" - "8510": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-8:9094" - "8511": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-9:9094" - "8512": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-10:9094" - "8513": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-11:9094" - "8514": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-12:9094" - "8515": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-13:9094" - "8516": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-14:9094" - "8517": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-15:9094" - "8518": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-16:9094" - "8519": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-17:9094" - "8520": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-18:9094" - "8521": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-19:9094" - "8522": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-20:9094" - "8523": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-21:9094" - "8524": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-22:9094" - "8525": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-23:9094" - "8526": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-24:9094" - "8527": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-25:9094" - "8528": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-26:9094" - "8529": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-27:9094" - "8530": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-28:9094" - "8531": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-29:9094" - "8532": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-30:9094" - "8533": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-31:9094" - "8534": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-32:9094" - "8535": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-33:9094" - "8536": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-34:9094" - "8537": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-35:9094" - "8538": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-36:9094" - "8539": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-37:9094" - "8540": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-38:9094" - "8541": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-39:9094" - "8542": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-40:9094" - "8543": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-41:9094" - "8544": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-42:9094" - "8545": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-43:9094" - "8546": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-44:9094" - "8547": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-45:9094" - "8548": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-46:9094" - "8549": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-47:9094" - "8550": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-48:9094" - "8551": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-49:9094" - "8552": "prd-backend-fulfillment-2/backend-fulfillment-2-broker-external4-50:9094" - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 5 - maxReplicas: 100 - targetCPUUtilizationPercentage: 40 - extraArgs: - enable-ssl-passthrough: true - ingressClassResource: - name: nginx-supply - service: - type: LoadBalancer - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-supply-tcp"}}}' - networking.gke.io/load-balancer-type: Internal - networking.gke.io/internal-load-balancer-allow-global-access: "true" - nodeSelector: - dedicated: nginx-supply - tolerations: - - key: "dedicated" - operator: "Equal" - value: "nginx-supply" - effect: "NoSchedule" - resources: - limits: - cpu: 2000m - memory: 2000Mi - requests: - cpu: 1000m - memory: 1500Mi - autoscaling: - enabled: true - minReplicas: 5 - maxReplicas: 100 diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index 797b8cf..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,496 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 200m - # memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-admin-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "central" - team: "mqkafka-sre" - service: "node-exporter-mqkafka-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-agent/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index 9d2f8cb..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,295 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-mqkafka-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-mqkafka-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-mq-spsre-vmagent-prd@meesho-mq-prd-0225.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-mqkafka.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-mqkafka-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "central" - team: "mqkafka-sre" - service: "vmagent-mqkafka-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-mqkafka-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-mqkafka-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 2 - memory: 4Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "victoriametrics" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-alert/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-alert/custom-values.yaml deleted file mode 100644 index c34dc35..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-alert/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-mqkafka-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-mqkafka-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-mqkafka-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-mqkafka-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.meeshogcp.in" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/central/open-source-kafka/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-mqkafka-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-mqkafka-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - dedicated: "vminsert" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-insert/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index 9917531..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,234 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-mqkafka-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-mqkafka-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "central" - team: "mqkafka-sre" - service: "vminsert-mqkafka-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 2 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vminsert" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 2 - memory: 2Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-mqkafka-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-select/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index 9acab63..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,295 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-mqkafka-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - extraVMSelects: [] - #extraVMSelects: - #- --storageNode=http://vmselect-mqkafka-di-prd.meeshogcp.in - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-mqkafka-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 96 - clusternative.maxConcurrentRequests: 256 - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "central" - team: "mqkafka-sre" - service: "vmselect-mqkafka-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 20 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "victoriametrics" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 1200m - memory: 4Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-mqkafka-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-storage/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index f8ff468..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,316 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-mqkafka-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "victoriametrics" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 200Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "central" - team: "mqkafka-sre" - service: "vmstorage-mqkafka-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 2 - memory: 7Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.91.3-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-demand-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/victoriametrics-agent/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index 60e1065..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,372 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "mqkafka-scrape.yaml" - - - -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - - -replicaCount: 2 -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - - - -fullnameOverride: "vm-agent-mqkafka-prd" -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-mqkafka-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - spec: - minReadySeconds: 1200 - progressDeadlineSeconds: 1800 - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.133.0 # rewrites Chart.AppVersion - variant: "" - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-mq-spsre-vmagent-prd@meesho-mq-prd-0225.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - - name: - # -- mount API token to pod directly - automountToken: true - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWrite: - # - https://vminsert-prd-mqkafka.meeshogcp.in/insert/100/prometheus/api/v1/write - - url: http://vm-insert-mqkafka-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "central" - team: "mqkafka-sre" - service: "vm-agent-mqkafka-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: - bu: "central" - team: "mqkafka-sre" - service: "vm-agent-mqkafka-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-mqkafka-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: false - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-mqkafka-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 5 - memory: 11Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "victoriametrics" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - -# -- Pod topologySpreadConstraints -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] -allowedMetricsEndpoints: - - /metrics - -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/victoriametrics-insert/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/victoriametrics-insert/custom-values.yaml deleted file mode 100644 index 26451c4..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/victoriametrics-insert/custom-values.yaml +++ /dev/null @@ -1,352 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - - - -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-mqkafka-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.107.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - lifecycle: {} - ports: - # -- VMInsert http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-mqkafka-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-mqkafka-prd-0.vm-storage-mqkafka-prd.victoriametrics.svc:8400" - - "vm-storage-mqkafka-prd-1.vm-storage-mqkafka-prd.victoriametrics.svc:8400" - - "vm-storage-mqkafka-prd-2.vm-storage-mqkafka-prd.victoriametrics.svc:8400" - - "vm-storage-mqkafka-prd-3.vm-storage-mqkafka-prd.victoriametrics.svc:8400" - - "vm-storage-mqkafka-prd-4.vm-storage-mqkafka-prd.victoriametrics.svc:8400" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "central" - team: "mqkafka-sre" - service: "vm-insert-mqkafka-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Pod's additional labels - podLabels: - bu: "central" - team: "mqkafka-sre" - service: "vm-insert-mqkafka-prd" - env: "prd" - priority: "p0" - type: "vminsert" - terminationGracePeriodSeconds: 30 - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 2 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vminsert" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 2 - memory: 2Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) for details - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-mqkafka-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/victoriametrics-select/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/victoriametrics-select/custom-values.yaml deleted file mode 100644 index 0890fc9..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/victoriametrics-select/custom-values.yaml +++ /dev/null @@ -1,409 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: false - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - externalService: - enabled: true - name: vmselect-mqkafka-prd-proxy - - # -- Enable clusternative service for LoadBalancer vmselect component - clusternativeService: - enabled: false - - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - extraVMSelects: [] - #extraVMSelects: - #- --storageNode=http://vmselect-mqkafka-di-prd.meeshogcp.in - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-mqkafka-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - terminationGracePeriodSeconds: 60 - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: true - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8481 - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-mqkafka-prd-0.vm-storage-mqkafka-prd.victoriametrics.svc:8401" - - "vm-storage-mqkafka-prd-1.vm-storage-mqkafka-prd.victoriametrics.svc:8401" - - "vm-storage-mqkafka-prd-2.vm-storage-mqkafka-prd.victoriametrics.svc:8401" - - "vm-storage-mqkafka-prd-3.vm-storage-mqkafka-prd.victoriametrics.svc:8401" - - "vm-storage-mqkafka-prd-4.vm-storage-mqkafka-prd.victoriametrics.svc:8401" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "central" - team: "mqkafka-sre" - service: "vm-select-mqkafka-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - # -- Pod's additional labels - podLabels: - bu: "central" - team: "mqkafka-sre" - service: "vm-select-mqkafka-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 20 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "victoriametrics" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 1200m - memory: 4Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-mqkafka-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - route: - # -- Enable deployment of HTTPRoute for select component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/select" .Values.vmselect }}' - # -- vmselect mode: deployment, daemonSet - mode: deployment - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-central-mqkafka-prd-ase1/victoriametrics-storage/custom-values.yaml b/helm-overrides/k8s-central-mqkafka-prd-ase1/victoriametrics-storage/custom-values.yaml deleted file mode 100644 index 3c78afa..0000000 --- a/helm-overrides/k8s-central-mqkafka-prd-ase1/victoriametrics-storage/custom-values.yaml +++ /dev/null @@ -1,393 +0,0 @@ -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - - - -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: false - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - lifecycle: {} - ports: - # -- VMStorage http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-mqkafka-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - envFrom: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage-v1" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage-v1" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 200Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "central" - team: "mqkafka-sre" - service: "vm-storage-mqkafka-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - podLabels: - bu: "central" - team: "mqkafka-sre" - service: "vm-storage-mqkafka-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 3 - memory: 22Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.91.3-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-demand-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-central-prd-ase1-c/contour-cert-checker/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/contour-cert-checker/custom-values.yaml deleted file mode 100644 index 9008363..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "0 12 * * *" - args: ["--cluster=k8s-central-prd-ase1-c"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - dedicated: central-devops - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1-c/contour-external/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/contour-external/custom-values.yaml deleted file mode 100644 index c34cb83..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/contour-external/custom-values.yaml +++ /dev/null @@ -1,85 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external - nodeSelector: - dedicated: contour-external - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external - nodeSelector: - dedicated: contour-external - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "30" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-central-prd"}}}' - ports: - http: 80 - https: 443 - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-central-prd-ase1-c/contour-internal-0/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/contour-internal-0/custom-values.yaml deleted file mode 100644 index ff59600..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,89 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - image: - tag: 1.28.0-debian-11-r8 - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-central-c-prd"}}}' - ports: - http: 80 - https: 443 - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-central-prd-ase1-c/contour-internal-1/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/contour-internal-1/custom-values.yaml deleted file mode 100644 index 8aa79ed..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,90 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 360s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - image: - tag: 1.28.0-debian-11-r8 - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - logLevel: error - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int1-central-prd"}}}' - ports: - http: 80 - https: 443 - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-central-prd-ase1-c/coredns/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/coredns/custom-values.yaml deleted file mode 100644 index 16570c2..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/coredns/custom-values.yaml +++ /dev/null @@ -1,27 +0,0 @@ -replicaCount: 16 - -labels: - bu: central - team: central-devops - env: prd - -clusterIP: 10.137.104.2 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - -nodeSelector: - dedicated: central-devops - kubernetes.io/os: linux diff --git a/helm-overrides/k8s-central-prd-ase1-c/external-secrets/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/external-secrets/custom-values.yaml deleted file mode 100644 index 61baf20..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/external-secrets/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Additional service annotations - annotations: {} - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - nodeSelector: - dedicated: central-devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - nodeSelector: - dedicated: central-devops - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - nodeSelector: - dedicated: central-devops diff --git a/helm-overrides/k8s-central-prd-ase1-c/flagger/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/flagger/custom-values.yaml deleted file mode 100644 index 58317fc..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/flagger/custom-values.yaml +++ /dev/null @@ -1,68 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: hpa-changes-26 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - dedicated: central-devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: central-devops - bu: infra - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-central-rollout-service.prd-central-rollout-service.svc.cluster.local" - - -rolloutService: - enabled: true - apps: - prd-devops-node-app-consumers: true - prd-devops-node-app: true - prd-devops-golang-app: true - prd-devops-golang-app-test: true - prd-devops-golang-app-crons: true diff --git a/helm-overrides/k8s-central-prd-ase1-c/fluentd/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/fluentd/custom-values.yaml deleted file mode 100644 index 7d3fd8c..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/fluentd/custom-values.yaml +++ /dev/null @@ -1,719 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-cntr-cnsre-fluentd-prd@meesho-central-ase1c-prd-0225.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: central - team: sre - type: fluentd - service: fluentd-central-prd - priority: p0 - env: prd - zone_extended: ase1c - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: central - team: sre - type: fluentd - service: fluentd-central-prd - priority: p0 - env: prd - zone_extended: ase1c - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: [] -# - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-central-prd-ase1-c/keda/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/keda/custom-values.yaml deleted file mode 100644 index 23eb546..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/keda/custom-values.yaml +++ /dev/null @@ -1,35 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: central-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - podLabels: - bu: "central" - team: "central-devops" - metricsAdapter: - bu: "central" - team: "central-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 100m - memory: 200Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 50m - memory: 150Mi - requests: - cpu: 10m - memory: 80Mi \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1-c/kube-dns/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/kube-dns/custom-values.yaml deleted file mode 100644 index 18ecf5f..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.137.104.2"],"prd.mrouter.int.svc.cluster.local":["10.137.104.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.137.104.2"]} diff --git a/helm-overrides/k8s-central-prd-ase1-c/kube-events/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/kube-events/custom-values.yaml deleted file mode 100644 index 0dea214..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-central-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - dedicated: central-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: central-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: central-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: central-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: central-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: central-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-central-prd-ase1-c/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 97c1621..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-central-ase1c-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-central-ase1c-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "central" - team: "central-sre" - service: "kube-state-metrics-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "central-devops" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 23m - memory: 169Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-central-prd-ase1-c/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index 7bc6f6b..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,119 +0,0 @@ -fullnameOverride: opentelemetry-central-ase1c-prd - -mode: daemonset - -priorityClassName: "system-node-critical" - -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.prd.meesho.int - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external - - key: node_pool - operator: NotIn - values: - - np-central-default-prd-ase1 - -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 100m - memory: 124Mi - limits: - cpu: 200m - memory: 256Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "central" - team: "sre" - service: "opentelemetry-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - zone_extended: "ase1c" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1-c/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index 12066c1..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,498 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-admin-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - - -# Extra labels to be added to node exporter pods -podLabels: - bu: "central" - team: "central-sre" - service: "node-exporter-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - zone_extended: "ase1c" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-central-prd-ase1-c/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index 7b2c1b0..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-central-ase1c-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-central-ase1c-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "central" - team: "central-sre" - service: "stackdriver-exporter-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - zone_extended: "ase1c" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-central-ase1c-prd-0225" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,compute.googleapis.com/instance,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - dedicated: "vmselect" - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-cntr-cnsre-stackdriver-prd@meesho-central-ase1c-prd-0622.iam.gserviceaccount.com - } - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/k8s-central-prd-ase1-c/telegraf-operator/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/telegraf-operator/custom-values.yaml deleted file mode 100644 index b9ae2ea..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,151 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "central" - team: "central-sre" - service: "telegraf-operator-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - zone_extended: "ase1c" - -nodeSelector: - dedicated: "vmselect" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false diff --git a/helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-agent/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index f427b19..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,297 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-central-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-central-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-common-np-cntr-prd@meesho-central-ase1c-prd-0625.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - - https://vminsert-infra-prd.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vminsert-central-ase1c-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "central" - team: "central-sre" - service: "vmagent-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmagent" - zone_extended: "ase1c" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-central-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-central-ase1c-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 4 - memory: 3Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-alert/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-alert/custom-values.yaml deleted file mode 100644 index f1e105d..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-alert/custom-values.yaml +++ /dev/null @@ -1,305 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-central-ase1c-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-central-ase1c-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-central-ase1c-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-central-ase1c-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/central-ase1c/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-central-ase1c-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmalert" - zone_extended: "ase1c" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - dedicated: "vmselect" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-insert/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index be5efd2..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,235 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-central-ase1c-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-central-ase1c-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "central" - team: "central-sre" - service: "vminsert-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "vminsert" - zone_extended: "ase1c" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 2 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vminsert" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 3 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 3 - memory: 2048Mi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-central-ase1c-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-select/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index eab95da..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,294 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-central-ase1c-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-central-ase1c-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - clusternative.maxConcurrentRequests: "256" - dedup.minScrapeInterval: 60s - search.maxConcurrentRequests: "32" - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 180s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "central" - team: "central-sre" - service: "vmselect-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmselect" - zone_extended: "ase1c" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 15 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "vmselect" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 10 - memory: 10Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-central-ase1c-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-storage/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index ebe9ffb..0000000 --- a/helm-overrides/k8s-central-prd-ase1-c/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,317 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-central-ase1c-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 500Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "central" - team: "central-sre" - service: "vmstorage-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - zone_extended: "ase1c" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 8 - memory: 107Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.93.7-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-central-ase1c-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-central-prd-ase1/README.md b/helm-overrides/k8s-central-prd-ase1/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/k8s-central-prd-ase1/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1/ai-gateway/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/ai-gateway/custom-values.yaml deleted file mode 100644 index 8685e03..0000000 --- a/helm-overrides/k8s-central-prd-ase1/ai-gateway/custom-values.yaml +++ /dev/null @@ -1,287 +0,0 @@ -# Custom values for Bifrost (ai-gateway) - Meesho Production -# Usage: helm install bifrost ./helm-templates/bifrost/ -f ./helm-templates/bifrost/custom-values.yaml -n prd-ai-gateway - -# -- Deployment Configuration -- -replicaCount: 2 - -fullnameOverride: "prd-ai-gateway" - -image: - repository: docker.io/maximhq/bifrost - pullPolicy: IfNotPresent - tag: "v1.4.17" - -# -- Service Account -- -serviceAccount: - create: true - automount: true - annotations: - iam.gke.io/gcp-service-account: sa-dvops-ai-gateway@meesho-central-prd-0622.iam.gserviceaccount.com - name: "prd-ai-gateway" - -# -- Pod Metadata -- -deploymentLabels: - bu: central - env: prod - team: devops - priority: p1 - priority_v2: sp1 - primary_owner: deep.shah - secondary_owner: anupam.satsangi - service: ai-gateway - service_type: producer-httpstateless - -podLabels: - bu: central - env: prod - team: devops - priority: p1 - priority_v2: sp1 - primary_owner: deep.shah - secondary_owner: anupam.satsangi - service: ai-gateway - service_type: producer-httpstateless - -podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: "8080" - prometheus.io/scrape: "true" - telegraf.influxdata.com/class: infra - -# -- Security Context -- -podSecurityContext: - fsGroup: 65534 - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - -securityContext: - capabilities: - drop: - - ALL - readOnlyRootFilesystem: false - runAsNonRoot: true - runAsUser: 65534 - -# -- Service -- -service: - type: ClusterIP - port: 8080 - -# -- Contour HTTPProxy -- -# ingress.enabled=false disables Bifrost's official K8s Ingress -# httpProxy.enabled=true enables the Meesho Contour HTTPProxy templates -# HTTPProxy templates read from ingress.* for hosts, class, etc. -httpProxy: - enabled: true -createContourGateway: true -namespace: prd-ai-gateway -contourResponseTimeout: false -ingress: - enabled: false - ingressClassName: contour-internal-1 - servicePortNumber: 8080 - enableWebsocket: false - hosts: - - host: ai-gateway.prd.meesho.int - paths: - - path: / - pathType: ImplementationSpecific - - host: llm-gateway.prd.meesho.int - name: prd-llm-gateway-0 - intraName: prd-llm-gateway-intra-0 - paths: - - path: / - pathType: ImplementationSpecific - slowStart: - enabled: false - aggression: 1 - minPercent: 10 - window: 120s - -# -- Resources -- -resources: - limits: - cpu: "5" - memory: 25Gi - requests: - cpu: "4" - memory: 20Gi - -# -- Health Probes -- -livenessProbe: - httpGet: - path: /health - port: http - scheme: HTTP - initialDelaySeconds: 15 - periodSeconds: 5 - timeoutSeconds: 3 - failureThreshold: 5 - successThreshold: 1 - -readinessProbe: - httpGet: - path: /health - port: http - scheme: HTTP - initialDelaySeconds: 15 - periodSeconds: 5 - timeoutSeconds: 3 - failureThreshold: 5 - successThreshold: 1 - -# -- HPA (disabled - using KEDA) -- -autoscaling: - enabled: false - -# -- Scheduling -- -nodeSelector: - dedicated: megatetralite - -tolerations: - - key: dedicated - operator: Equal - value: megatetralite - effect: NoSchedule - -affinity: {} - -# -- Lifecycle & Graceful Shutdown -- -terminationGracePeriodSeconds: 300 -lifecycle: - preStop: - exec: - command: - - /bin/bash - - "-c" - - "kill -SIGQUIT; /bin/sleep 120" - -# -- Bifrost Application Config -- -bifrost: - appDir: /app/data - port: 8080 - host: 0.0.0.0 - logLevel: warn - logStyle: json - - # Auth configured via Bifrost UI (stored in DB), not in Helm values - # This avoids blocking /metrics scrape while still protecting the dashboard - - client: - dropExcessRequests: false - initialPoolSize: 300 - allowedOrigins: - - "*" - enableLogging: true - disableContentLogging: false - disableDbPingsInHealth: false - logRetentionDays: 365 - enforceGovernanceHeader: false - allowDirectKeys: false - maxRequestBodySizeMb: 100 - enableLitellmFallbacks: false - - # Configure providers with env.VAR_NAME references for API keys - # providers: - # openai: - # - keys: - # - value: "env.OPENAI_API_KEY" - # models: ["gpt-4o", "gpt-4o-mini"] - # weight: 1.0 - -# -- Storage (External PostgreSQL) -- -storage: - mode: postgres - configStore: - enabled: true - logsStore: - enabled: true - -postgresql: - enabled: false - external: - enabled: true - host: "10.147.2.236" - port: 5432 - user: "app_user_bifrost" - database: "bifrost_db" - sslMode: "disable" - existingSecret: "prd-ai-gateway-vault" - passwordKey: "BIFROST_POSTGRES_PASSWORD" - -# -- Vector Store (disabled) -- -vectorStore: - enabled: false - type: none - -# -- Meesho Standard Env Vars -- -env: - - name: TZ - value: "Asia/Kolkata" - - name: TELEGRAF_UDP_HOST - valueFrom: - fieldRef: - fieldPath: status.podIP - - name: NODE_IP - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - name: POD_NAME - valueFrom: - fieldRef: - fieldPath: metadata.name - - name: POD_NAMESPACE - valueFrom: - fieldRef: - fieldPath: metadata.namespace - - name: POD_IP - valueFrom: - fieldRef: - fieldPath: status.podIP - -# --- Meesho Infrastructure Extensions --- - -# -- PodDisruptionBudget -- -podDisruptionBudget: - enabled: true - maxUnavailable: "10%" - -# -- ExternalSecret (Vault) -- -# Creates K8s Secret "prd-ai-gateway-vault" from Vault path -# This secret is referenced by postgresql.external.existingSecret above -externalSecret: - enabled: true - secretName: "prd-ai-gateway-vault" - path: "prd/cntr/devop/ai-gateway" - refreshInterval: "0" - secretStoreRef: "vault-backend" - -# -- KEDA ScaledObject -- -keda: - enabled: true - pollingInterval: 30 - minReplicaCount: 2 - maxReplicaCount: 200 - scaledown: - stabilizationWindowSeconds: 1800 - selectpolicy: Min - policies: - - type: Pods - value: 2 - periodseconds: 15 - scaleup: - stabilizationWindowSeconds: 120 - selectpolicy: Max - policies: - - type: Pods - value: 2 - periodseconds: 15 - - type: Percent - value: 10 - periodseconds: 15 - triggers: - - type: cpu - metricType: Utilization - metadata: - value: "40" diff --git a/helm-overrides/k8s-central-prd-ase1/akamai-observability-mcp/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/akamai-observability-mcp/custom-values.yaml deleted file mode 100644 index 2ed313a..0000000 --- a/helm-overrides/k8s-central-prd-ase1/akamai-observability-mcp/custom-values.yaml +++ /dev/null @@ -1,99 +0,0 @@ -# Akamai Observability MCP — grafana-mcp chart on k8s-central-prd-ase1 -# Chart: helm-templates/grafana-mcp (v2.0.0+) - -fullnameOverride: "akamai-observability-mcp" - -replicas: 1 - -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/prd/devop/grafana-mcp - tag: "v2026-03-10" - pullPolicy: IfNotPresent - -labels: - bu: central - team: devops - service: akamai-observability-mcp - env: prd - -# -- Grafana connection. -# url: set this to the Akamai/observability Grafana endpoint reachable from -# k8s-central-prd-ase1 (in-cluster DNS preferred; otherwise the prd FQDN). -# apiKeySecret: read GRAFANA_SERVICE_ACCOUNT_TOKEN from the K8s Secret produced -# by the ExternalSecret below. -grafana: - url: "" # TODO: set the Grafana base URL (e.g. https://grafana-akamai.prd.meesho.int) - apiKeySecret: - name: "akamai-observability-mcp-vault" - key: "GRAFANA_SERVICE_ACCOUNT_TOKEN" - -# -- Vault-backed secret. Mint the Grafana service-account token in the Grafana -# UI, store it at the path below under key GRAFANA_SERVICE_ACCOUNT_TOKEN, then -# ESO syncs it into the K8s Secret referenced above. -externalSecret: - enabled: true - secretName: "akamai-observability-mcp-vault" - path: "prd/cntr/devop/akamai-observability-mcp" # TODO: confirm Vault path - refreshInterval: "0" - secretStoreRef: "vault-backend" - -serviceAccount: - create: true - annotations: {} - -# SSE transport is long-lived — keep Contour from cutting connections. -contourResponseTimeout: "1h" - -createContourGateway: true - -ingress: - enabled: true - ingressClassName: contour-internal-1 - servicePortNumber: 8000 - enableWebsocket: true - hosts: - - host: akamai-observability-mcp.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 - -resources: - requests: - cpu: 250m - memory: 256Mi - limits: - cpu: 500m - memory: 512Mi - -# Scheduling — dedicated MCP node pool on k8s-central-prd-ase1. -nodeSelector: - dedicated: "devops-mcp" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops-mcp" - effect: NoSchedule - -securityContext: - fsGroup: 1000 - runAsNonRoot: true - runAsUser: 1000 - runAsGroup: 1000 - -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - readOnlyRootFilesystem: true - runAsNonRoot: true - runAsUser: 1000 - runAsGroup: 1000 diff --git a/helm-overrides/k8s-central-prd-ase1/alloy/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/alloy/custom-values.yaml deleted file mode 100644 index bdc658e..0000000 --- a/helm-overrides/k8s-central-prd-ase1/alloy/custom-values.yaml +++ /dev/null @@ -1,53 +0,0 @@ -fullnameOverride: "alloy-central-prd" - -alloy: - configMap: - configFile: central.alloy - - clustering: - enabled: true - - extraPorts: - - name: "otlp-grpc" - port: 4317 - targetPort: 4317 - protocol: "TCP" - - name: "otlp-http" - port: 4318 - targetPort: 4318 - protocol: "TCP" - - resources: - requests: - cpu: 7 - memory: 55Gi - limits: - cpu: 7.5 - memory: 60Gi - -configReloader: - enabled: true - -serviceAccount: - annotations: { - iam.gke.io/gcp-service-account: sa-cntr-sre-grafna-obs-stk-prd@meesho-central-prd-0622.iam.gserviceaccount.com - } - -controller: - # Must be one of 'daemonset', 'deployment', or 'statefulset'. - type: 'deployment' - - nodeSelector: - dedicated: "alloy" - tolerations: - - key: "dedicated" - operator: "Equal" - value: "alloy" - effect: "NoSchedule" - - autoscaling: - enabled: true - minReplicas: 1 - maxReplicas: 50 - targetCPUUtilizationPercentage: 80 - targetMemoryUtilizationPercentage: 60 \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1/argocd/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/argocd/custom-values.yaml deleted file mode 100644 index 80a5314..0000000 --- a/helm-overrides/k8s-central-prd-ase1/argocd/custom-values.yaml +++ /dev/null @@ -1,130 +0,0 @@ -argo-cd: - global: - additionalLabels: - bu: central - team: central-devops - podLabels: - bu: central - team: central-devops - nodeSelector: - dedicated: devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - dex: - enabled: true - resources: - limits: - cpu: 250m - memory: 512Mi - requests: - cpu: 250m - memory: 512Mi - metrics: - enabled: true - controller: - replicas: 2 - enableStatefulSet: true - resources: - limits: - cpu: "1" - memory: 2048Mi - requests: - cpu: 500m - memory: 1024Mi - redis-ha: - enabled: true - repoServer: - autoscaling: - enabled: true - maxReplicas: 10 - minReplicas: 2 - metrics: - enabled: true - serviceMonitor: - enabled: false - interval: 60s - resources: - limits: - cpu: 1500m - memory: 2Gi - requests: - cpu: "1" - memory: 1Gi - server: - replicas: 3 - autoscaling: - enabled: false - minReplicas: 3 - maxReplicas: 10 - extraArgs: - - --insecure - ingress: - annotations.nginx.ingress.kubernetes.io/force-ssl-redirect: false - annotations.nginx.ingress.kubernetes.io/rewrite-target: / - annotations.nginx.ingress.kubernetes.io/ssl-redirect: false - enabled: true - hosts: - - argocd-central-prd.meeshogcp.in - ingressClassName: contour-internal - resources: - limits: - cpu: "1" - memory: 2048Mi - requests: - cpu: 500m - memory: 1024Mi - config: - url: https://argocd-central-prd.meeshogcp.in - statusbadge.enabled: "true" - help.chatUrl: "https://meesho.slack.com/archives/C021QNS6JLV" - help.chatText: "Chat now!" - dex.config: | - logger: - level: error - format: json - connectors: - - type: github - id: github - name: GitHub - loadAllGroups: true - admin.enabled: "true" - config: - clientID: 281775fcf1df0edda53c - clientSecret: $github-sso-secret:dex.github.clientSecret - orgs: - - name: Meesho - rbacConfig: - policy.csv: | - p, role:admins, *, *, */*, allow - g, Meesho:devops, role:admin - p, role:backend, applications, create, cntr-*/*, allow - p, role:backend, applications, get, cntr-*/*, allow - p, role:backend, applications, override, cntr-*/*, allow - p, role:backend, applications, sync, cntr-*/*, allow - p, role:backend, applications, update, cntr-*/*, allow - p, role:backend, applications, delete, cntr-*/*, allow - p, role:backend, logs, get, cntr-*/*, allow - p, role:backend, exec, create, cntr-*/*, allow - p, role:backend, projects, get, cntr-*, allow - p, role:backend, projects, sync, cntr-*, allow - p, role:backend, applications, action/apps/Deployment/restart, cntr-*/*, allow - p, role:backend, repositories, update, cntr-*/*, allow - g, Meesho:backend, role:backend - g, ringmaster, role:admins - applicationSet: - replicaCount: 2 - notifications: - metrics: - enabled: true - serviceMonitor: - enabled: false - resources: - limits: - cpu: 500m - memory: 1Gi - requests: - cpu: 300m - memory: 512Mi diff --git a/helm-overrides/k8s-central-prd-ase1/aurva-dataplane/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/aurva-dataplane/custom-values.yaml deleted file mode 100644 index f9a5e7a..0000000 --- a/helm-overrides/k8s-central-prd-ase1/aurva-dataplane/custom-values.yaml +++ /dev/null @@ -1,675 +0,0 @@ -## Aurva Data Plane -## Ref: https://github.com/aurva-io/aurva-charts.git - -postgresql: - enabled: true - fullnameOverride: "aurva-dataplane-database" - volumePermissions: - ## @param volumePermissions.enabled Enable init container that changes the owner and group of the persistent volume - ## - enabled: true - global: - storageClass: pd-standard-retain - postgresql: - auth: - postgresPassword: "aurva" - database: "controller" - # Add toleration to make sure where this postgres db pod should reside (Applicable for production workloads): For more detail ref: https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/ - primary: - extendedConfiguration: | - max_connections = 300 - - #PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - - # -- Select nodes to deploy which matches the following labels - nodeSelector: ##PLACEHOLDER## - dedicated: central-devops - -# -- Provide a name in place of `aurva` -# namespaceOverride: aurva-dataplane - -########################################################## -# Global Configs -########################################################## -global: - aurva_controller: - enabled: true - aurva_fastdet: - enabled: false - aurva_pii_analyzer: - enabled: true - aurva_ocr: - enabled: true - aurva_collector: - enabled: true - - deploymentAnnotations: {} - - priorityClassName: "" - -########################################################## -# Aurva Controller -########################################################## -aurva_controller: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "central" - team: "central-devops" - service: "aurva-central-prd" - env: "prd" - priority: "p0" - type: "aurva_controller" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 10 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - image: - # Image of the app container - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-controller/aurva-controller - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - - type: secret - name: aurva-controller-secrets - # - type: configmap - # name: proxy-datasource-config - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 2 - requests: - memory: 1Gi - cpu: 1 - - aurvaFastdet: - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-fastdet/aurva-fastdet - tag: "v2.30.13" - pullPolicy: IfNotPresent - envFrom: [] - env: [] - resources: - limits: - cpu: 0.5 - memory: 512Mi - requests: - cpu: 0.5 - memory: 512Mi - - nodeSelector: ##PLACEHOLDER## - dedicated: central-devops - - # -- Taint tolerations for nodes - ##PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-controller-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - config: - #variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - COMMAND_URL: "command.aurva-prd.meeshogcp.in:80" - DEPLOYMENT_TYPE: "kubernetes" - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - FLUSHER_WORKER_POOL_SIZE: "1000" - FLUSHER_BATCH_SIZE: "30000" - UNIQUENESS_IDENTIFIER: "k8s-central-prd-ase1" #Recommendation: should be equal to cluster name - PROVIDER_ACCOUNT_ID: "meesho-central-prd-0622" # GCP PROJECT ID (not Number) - REGION: "asia-southeast1" #eg: asia-south1 - ENVIRONMENT: "prod" - OCR_ENABLED: "true" - MONITORING_ENABLED: "false" - HYBRID_ONLY_MODE: "false" - FORCE_TLS: "false" - FASTDET_FLAG : "true" - AADHAAR_ENHANCER: "0" - WORKSPACE_EVENT_TRACKING_ENABLED: "false" - ACCESS_IQ_ENABLED: "true" - GRPC_ENFORCE_ALPN_ENABLED: "false" - ENABLE_FASTDET: "true" - FASTDET_MAX_BATCH_SIZE: "100" - HEARTBEAT_INTERVAL: "5m" - #pii data - PII_BUCKET_NAME: "gcs-infra-devop-aurva-central-prd" - PII_LOG_BUCKET_REGION: "asia-southeast1" - PII_LOG_CRON: "*/5 * * * *" - ENABLE_PII_LOG: "true" - PII_EVIDENCE_UPLOAD_MAX_BLOCKING_TASKS: "50000" - PII_EVIDENCE_UPLOAD_MAX_CONCURRENT_TASKS: "500" - PII_EVIDENCE_MAX_CACHE_WEIGHT: "100" - QUOTA_CLEANUP_CRON: "0 0 * * *" - ENABLE_PII_QUOTA: "true" - MAX_PII_EVIDENCES_PER_KEY_PER_WINDOW: "3" - PII_QUOTA_SYNC_CRON: "*/2 * * * *" - QUOTA_WINDOW_HOURS: "24" - #constants - SKIP_NAMESPACES: "contour-internal-1-central-prd,contour-internal-0-central-prd,contour-external-central-prd" - CLOUD_PROVIDER: "gcp" - LOG_ENV: "production" - RDS_SCANNER_AVAILABILITY : "false" - REDSHIFT_SCANNER_AVAILABILITY : "false" - S3_SCANNER_AVAILABILITY : "false" - DYNAMO_SCANNER_AVAILABILITY: "false" - DOCDB_SCANNER_AVAILABILITY: "false" - OPENSEARCH_SCANNER_AVAILABILITY: "false" - CLOUDSQL_SCANNER_AVAILABILITY: "true" - BIGQUERY_SCANNER_AVAILABILITY: "true" - AWS_SNAPSHOT_SCANNER_AVAILABILITY: "false" - CLOUDSTORAGE_SCANNER_AVAILABILITY: "true" - KEYSPACES_SCANNER_AVAILABILITY: "false" - ALLOYDB_SCANNER_AVAILABILITY: "true" - BIGTABLE_SCANNER_AVAILABILITY: "true" - GCP_BACKUP_AVAILABILITY: "true" - EGRESS_MODE_ONLY: "false" - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - SCAN_UUID_ENABLED: "true" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-controller-sa - # -- Annotations applied to created service account - annotations: - iam.gke.io/gcp-service-account: sa-central-prd-aurva-contr@meesho-central-prd-0622.iam.gserviceaccount.com -# eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 15 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 - - -########################################################## -# Aurva OCR -########################################################## -aurva_ocr: - # -- Additional labels for aurva-controller - additionalLabels: - bu: "central" - team: "central-devops" - service: "aurva-central-prd" - env: "prd" - priority: "p0" - type: "aurva_ocr" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 1 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-ocr/aurva-ocr - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-ocr: - type: secret - name: aurva-ocr-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 1 - requests: - memory: 2Gi - cpu: 1 - - # -- Select nodes to deploy which matches the following labels - - nodeSelector: ##PLACEHOLDER## - dedicated: central-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-ocr-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - OCR_TIME_LIMIT: "1" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-central-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-ocr-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # iam.gke.io/gcp-service-account: service-account@gcp.iam.gserviceaccount.com - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva Collector -########################################################## -aurva_collector: - - # -- Additional labels for aurva-analyzer - additionalLabels: - bu: "central" - team: "central-devops" - service: "aurva-central-prd" - env: "prd" - priority: "p0" - type: "aurva_collector" - - - # -- Annotations on aurva-analyzer - annotations: {} -# "key": "value" - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-collector/aurva-collector - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-controller: - type: secret - name: aurva-collector-secrets - - resources: - limits: - cpu: 800m - memory: 800Mi - requests: - cpu: 200m - memory: 512Mi - - podSecurityContext: {} - - securityContext: - privileged: true - capabilities: - add: - # For kernel v5.8 and above we don't need CAP_SYS_ADMIN or CAP_SYS_RESOURCE - # we just need CAP_BPF and CAP_PERFMON. This has been tested on our EKS node - # which is on kernel v5.10.x - # When SSL Tracing is required we need CAP_SYS_ADMIN and CAP_SYS_PTRACE - # on top of the previous capabilities - # So finally these are the 4 possible combinations for capabilies - # 1. Newer Kernels without SSL - # - BPF - # - PERFMON - # 2. Newer Kernels with SSL - - SYS_ADMIN - - SYS_PTRACE - # 3. Older Kernels without SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # 4. Older Kernels with SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # - SYS_PTRACE - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - volumes: - - name: debugfs - mountPath: /sys/kernel/debug - hostPath: /sys/kernel/debug - - name: vmlinux - mountPath: /sys/kernel/btf/vmlinux - hostPath: /sys/kernel/btf/vmlinux - - name: procfs - mountPath: /host/proc - hostPath: /proc - - name: bpffs - mountPath: /sys/fs/bpf - hostPath: /sys/fs/bpf - - # -- Taint tolerations for nodes - tolerations: - # - effect: NoSchedule - # key: dedicated - # operator: Equal - # value: megatetra - - operator: Exists - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external - - alloy - - preprod-spot-16 - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-collector-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - # variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-central-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - TRACE_INTERNAL_SVC: "true" - TRACE_INTERNAL_SVC_HTTP: "true" - INTERNAL_SVC_SAMPLE_INTERVAL: "10m" - LOGS_TTL: "1h" - TRACE_HTTP2: "true" - TRACE_SSL: "false" - TRACE_PSQL: "false" - TRACE_SQLSERVER: "false" - TRACE_MYSQL: "false" - TRACE_EGRESS: "true" - TRACE_GO_TLS: "false" - TRACE_ML_SERVICES: "false" - # constants - LOG_ENV: production - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - MONITORING_ENABLED: "false" - ENABLE_INGRESS_INFORMER: "false" - ENABLE_SERVICE_INFORMER: "false" - ENABLE_ISTIO_INFORMER: "false" - EXCLUDED_PII_REGEX_TYPES: "ip_address,us_bank_number,us_driver_license,us_itin,us_passport,us_routing,us_mbi,ssn" - AGGREGATOR_MAX_CONNECTIONS: "1000" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-collector-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva PII Analyzer -########################################################## -aurva_pii_analyzer: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "central" - team: "central-devops" - service: "aurva-central-prd" - env: "prd" - priority: "p0" - type: "aurva_pii_analyzer" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 3 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - ##PLACEHOLDER## - nodeSelector: - dedicated: central-devops - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-piianalyzer/aurva-piianalyzer - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-pii-analyzer: - type: secret - name: aurva-pii-analyzer-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 4Gi - cpu: 4 - requests: - memory: 2Gi - cpu: 2 - - nodeSelector: ##PLACEHOLDER## - dedicated: central-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-pii-analyzer-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - SCHEDULER_TIME: "1" - SUPPORTED_REGION: "US" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-central-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-pii-analyzer-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 3 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 diff --git a/helm-overrides/k8s-central-prd-ase1/cert-manager/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/cert-manager/custom-values.yaml deleted file mode 100644 index b5a12a1..0000000 --- a/helm-overrides/k8s-central-prd-ase1/cert-manager/custom-values.yaml +++ /dev/null @@ -1,129 +0,0 @@ -global: - logLevel: 2 - rbac: - create: true - priorityClassName: "high-priority" -installCRDs: false - -crds: - enabled: true - keep: true - -# Cert-manager Controller -replicaCount: 1 -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-controller - tag: v1.20.1 - pullPolicy: IfNotPresent - -nodeSelector: - dedicated: central-devops - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - -resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 500m - memory: 512Mi - -# Webhook Configuration -webhook: - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-webhook - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: central-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# CA Injector Configuration -cainjector: - enabled: true - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-cainjector - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: central-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# ACME Solver Configuration -acmesolver: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-acmesolver - tag: v1.20.1 - pullPolicy: IfNotPresent - -# Startup API Check -startupapicheck: - enabled: true - timeout: 1m - backoffLimit: 4 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-startupapicheck - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: central-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - -# Prometheus Monitoring -prometheus: - enabled: true - servicemonitor: - enabled: false - interval: 60s - scrapeTimeout: 30s - labels: - prometheus: cert-manager diff --git a/helm-overrides/k8s-central-prd-ase1/clickhouse/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/clickhouse/custom-values.yaml deleted file mode 100644 index 2dc5996..0000000 --- a/helm-overrides/k8s-central-prd-ase1/clickhouse/custom-values.yaml +++ /dev/null @@ -1,251 +0,0 @@ -# logHouse (central-prd) - overrides only. -# Google SSO via oauth2-proxy; ClickHouse ingress disabled. -oauth2Proxy: - enabled: true - -# nginx audit proxy maps X-Forwarded-Email → X-ClickHouse-Setting-log_comment -# so system.query_log.log_comment shows the SSO email of who ran each query. -auditProxy: - enabled: true - replicas: 2 - image: "nginx:1.27-alpine" - tolerations: - - key: "dedicated" - operator: "Equal" - value: "loghouse" - effect: "NoSchedule" - -externalSecret: - enabled: true - path: meesho/prd/cntr/devop/loghouse - secretName: loghouse-oauth2-secret - annotations: {} - -# --- Bitnami ClickHouse subchart --- -clickhouse: - replicaCount: 3 - global: - security: - allowInsecureImages: true - - image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/prd/sis/clickhouse - tag: 25.6.2-debian-12-r0 - - auth: - username: default - password: "" - existingSecret: "loghouse-oauth2-secret" - existingSecretKey: "clickhouse-password" - - - # Enable sampling so Bitnami's 08-sampling.xml preserves query_log, - # text_log, metric_log etc. All queries are recorded in system.query_log. - sampling: - enabled: true - - usersdFiles: - grant_all.xml: | - - - - 1 - 1 - - - - log_queries.xml: | - - - - 1 - 0 - - - - - initContainers: - - name: copy-usersd-config - image: busybox:1.36 - command: - - /bin/sh - - -ec - - cp -R /src/. /dst/ - volumeMounts: - - name: usersd-configuration-configuration - mountPath: /src - readOnly: true - - name: clickhouse-users-d - mountPath: /dst - - persistence: - storageClass: "premium-rwo" - size: 80Gi - mountPath: /var/lib/clickhouse - - - extraEnvVars: - - name: CLICKHOUSE_USER - value: "default" - - name: CLICKHOUSE_PASSWORD - valueFrom: - secretKeyRef: - name: loghouse-oauth2-secret - key: clickhouse-password - extraVolumes: - - name: clickhouse-users-d - emptyDir: - sizeLimit: 100Mi - - name: clickhouse-logs - emptyDir: - sizeLimit: 500Mi - - name: fluentbit-config - configMap: - name: loghouse-fluentbit-config - extraVolumeMounts: - - name: clickhouse-users-d - mountPath: /etc/clickhouse-server/users.d - - name: clickhouse-logs - mountPath: /var/log/clickhouse-server - - sidecars: - - name: query-log-tailer - image: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/prd/sis/clickhouse:25.6.2-debian-12-r0" - command: - - /bin/sh - - -c - - | - while true; do - clickhouse-client --host 127.0.0.1 --port 9000 --user default --password "$CLICKHOUSE_PASSWORD" --query="SELECT event_time, user, query_id, query, client_hostname FROM system.query_log WHERE type = 'QueryFinish' AND event_time > now() - INTERVAL 10 SECOND FORMAT JSONEachRow" 2>/dev/null; - sleep 10; - done - env: - - name: CLICKHOUSE_PASSWORD - valueFrom: - secretKeyRef: - name: loghouse-oauth2-secret - key: clickhouse-password - volumeMounts: - - name: clickhouse-logs - mountPath: /var/log/clickhouse-server - readOnly: true - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - - name: fluentbit - image: fluent/fluent-bit:3.1 - resources: - requests: - cpu: 25m - memory: 50Mi - limits: - cpu: 100m - memory: 100Mi - volumeMounts: - - name: clickhouse-logs - mountPath: /var/log/clickhouse-server - readOnly: true - - name: fluentbit-config - mountPath: /fluent-bit/etc - readOnly: true - - defaultInitContainers: - volumePermissions: - enabled: false - image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/prd/sis/os-shell - tag: 12-debian-12-r47 - - resourcesPreset: "none" - # Chart maps these inversely: values.requests -> pod limits, values.limits -> pod requests - resources: - requests: - cpu: "6" - memory: 40Gi - limits: - cpu: "6" - memory: 40Gi - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "loghouse" - effect: "NoSchedule" - - # Disabled when oauth2Proxy.enabled is true (oauth2-proxy handles ingress) - ingress: - enabled: false - - networkPolicy: - enabled: true - allowExternal: true - allowExternalEgress: true - - keeper: - enabled: false - - -oauth2-proxy: - replicaCount: 3 - config: - existingSecret: loghouse-oauth2-secret - requiredSecretKeys: - - client-id - - client-secret - - cookie-secret - extraArgs: - provider: google - redirect-url: "http://loghouse-central.prd.meesho.int/oauth2/callback" - upstream: "http://loghouse-central-prd-audit-proxy:8123" - email-domain: "meesho.com" - proxy-prefix: "/oauth2" - pass-host-header: "true" - proxy-websockets: "true" - real-client-ip-header: "X-Forwarded-For" - cookie-secure: "false" - cookie-expire: "0s" - custom-templates-dir: "/templates" - skip-jwt-bearer-tokens: "true" - oidc-issuer-url: "https://accounts.google.com" - extra-jwt-issuers: "https://accounts.google.com=32555940559.apps.googleusercontent.com" - pass-user-headers: "true" - set-xauthrequest: "true" - request-logging: "true" - auth-logging: "true" - standard-logging: "true" - extraVolumes: - - name: custom-templates - configMap: - name: '{{ .Release.Name }}-oauth2-proxy-templates' - extraVolumeMounts: - - name: custom-templates - mountPath: /templates - readOnly: true - service: - portNumber: 80 - ingress: - enabled: true - className: contour-internal-1 - path: / - pathType: Prefix - hosts: - - loghouse-central.prd.meesho.int - annotations: {} - tls: [] - sessionStorage: - type: cookie - redis-ha: - enabled: false - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 200m - memory: 128Mi diff --git a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-external-arm.yaml b/helm-overrides/k8s-central-prd-ase1/computeclass/contour-external-arm.yaml deleted file mode 100644 index a6f2575..0000000 --- a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-external-arm.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-external-arm -spec: - nodePoolConfig: - serviceAccount: sa-cntr-cndvs-dvps-v131-prd@meesho-central-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-external-cc.yaml b/helm-overrides/k8s-central-prd-ase1/computeclass/contour-external-cc.yaml deleted file mode 100644 index 7b883b8..0000000 --- a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-external-cc.yaml +++ /dev/null @@ -1,34 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-external-cc -spec: - nodePoolConfig: - serviceAccount: sa-cntr-cndvs-dvps-v131-prd@meesho-central-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - spot: false - maxPodsPerNode: 16 - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: n2d-highcpu-8 - spot: false - maxPodsPerNode: 16 - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-0-arm.yaml b/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-0-arm.yaml deleted file mode 100644 index 84736c1..0000000 --- a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-0-arm.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-0-arm -spec: - nodePoolConfig: - serviceAccount: sa-cntr-cndvs-dvps-v131-prd@meesho-central-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-0-cc.yaml b/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-0-cc.yaml deleted file mode 100644 index c8d9b01..0000000 --- a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-0-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-0-cc -spec: - nodePoolConfig: - serviceAccount: sa-cntr-cndvs-dvps-v131-prd@meesho-central-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-1-arm.yaml b/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-1-arm.yaml deleted file mode 100644 index b2ac9a9..0000000 --- a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-1-arm.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-1-arm -spec: - nodePoolConfig: - serviceAccount: sa-cntr-cndvs-dvps-v131-prd@meesho-central-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4a-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: n4a-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-1-cc.yaml b/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-1-cc.yaml deleted file mode 100644 index 5a45656..0000000 --- a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-1-cc.yaml +++ /dev/null @@ -1,34 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-1-cc -spec: - nodePoolConfig: - serviceAccount: sa-cntr-cndvs-dvps-v131-prd@meesho-central-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - spot: false - maxPodsPerNode: 16 - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: n2d-highcpu-8 - spot: false - maxPodsPerNode: 16 - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-intra-0-arm.yaml b/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-intra-0-arm.yaml deleted file mode 100644 index 805a5f6..0000000 --- a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-intra-0-arm.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-0-arm -spec: - nodePoolConfig: - serviceAccount: sa-cntr-cndvs-dvps-v131-prd@meesho-central-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-intra-0-cc.yaml b/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-intra-0-cc.yaml deleted file mode 100644 index a91f46f..0000000 --- a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-intra-0-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-0-cc -spec: - nodePoolConfig: - serviceAccount: sa-cntr-cndvs-dvps-v131-prd@meesho-central-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-intra-1-arm.yaml b/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-intra-1-arm.yaml deleted file mode 100644 index eb94e4f..0000000 --- a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-intra-1-arm.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-1-arm -spec: - nodePoolConfig: - serviceAccount: sa-cntr-cndvs-dvps-v131-prd@meesho-central-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-intra-1-cc.yaml b/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-intra-1-cc.yaml deleted file mode 100644 index ee436b4..0000000 --- a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-internal-intra-1-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-intra-1-cc -spec: - nodePoolConfig: - serviceAccount: sa-cntr-cndvs-dvps-v131-prd@meesho-central-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-shared-arm.yaml b/helm-overrides/k8s-central-prd-ase1/computeclass/contour-shared-arm.yaml deleted file mode 100644 index c14aca9..0000000 --- a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-shared-arm.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-shared-arm -spec: - nodePoolConfig: - serviceAccount: sa-cntr-cndvs-dvps-v131-prd@meesho-central-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-shared-cc.yaml b/helm-overrides/k8s-central-prd-ase1/computeclass/contour-shared-cc.yaml deleted file mode 100644 index ff9d3c9..0000000 --- a/helm-overrides/k8s-central-prd-ase1/computeclass/contour-shared-cc.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-shared-cc -spec: - nodePoolConfig: - serviceAccount: sa-cntr-cndvs-dvps-v131-prd@meesho-central-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: false - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n2d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1/conntrack-adjuster/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index ebd3809..0000000 --- a/helm-overrides/k8s-central-prd-ase1/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,25 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - max: 2097152 - hashsize: 524288 - sleepInterval: 30 - -nodeAffinity: - enabled: true - nodeSelectorTerms: - - matchExpressions: - - key: cloud.google.com/compute-class - operator: In - values: - - "contour-external-cc" - - "contour-internal-0-cc" - - "contour-internal-1-cc" - - "contour-intra-0-cc" - - "contour-intra-1-cc" - - matchExpressions: - - key: dedicated - operator: In - values: - - "contour-external-1" diff --git a/helm-overrides/k8s-central-prd-ase1/contour-ca-issuer/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/contour-ca-issuer/custom-values.yaml deleted file mode 100644 index 9212add..0000000 --- a/helm-overrides/k8s-central-prd-ase1/contour-ca-issuer/custom-values.yaml +++ /dev/null @@ -1,17 +0,0 @@ -# ClusterIssuer override for k8s-demand-prd-ase1 (prd demand cluster, ase1a zone). -# ClusterIssuer is a cluster-scoped resource — this file pins the issuer -# identity for this cluster so the name is auditable per-cluster. -# -# Must match the `issuerRef.name` in consuming Certificate CRs (see -# devops-helm-charts/2.0.0/templates/proxyless-grpc-cert.yaml). - -issuerName: contour-central-prd-ca-issuer -rootCASecretName: contour-central-ca - -externalSecret: - enabled: true - vaultPath: meesho/devops/contour/root-ca - refreshInterval: "0" - secretStoreRef: - name: vault-backend -namespace: cert-manager-central-prd diff --git a/helm-overrides/k8s-central-prd-ase1/contour-cert-checker/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/contour-cert-checker/custom-values.yaml deleted file mode 100644 index c1e9192..0000000 --- a/helm-overrides/k8s-central-prd-ase1/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "0 12 * * *" - args: ["--cluster=k8s-central-prd-ase1"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - dedicated: central-devops - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1/contour-external-1/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/contour-external-1/custom-values.yaml deleted file mode 100644 index 288f8f5..0000000 --- a/helm-overrides/k8s-central-prd-ase1/contour-external-1/custom-values.yaml +++ /dev/null @@ -1,101 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external-1 - nodeSelector: - dedicated: contour-external-1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external-1 - nodeSelector: - dedicated: contour-external-1 - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-1-central-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-central-prd-ase1/contour-external/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/contour-external/custom-values.yaml deleted file mode 100644 index 88a9e09..0000000 --- a/helm-overrides/k8s-central-prd-ase1/contour-external/custom-values.yaml +++ /dev/null @@ -1,101 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-external-arm - nodeSelector: - cloud.google.com/compute-class: contour-external-arm - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-central-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-central-prd-ase1/contour-internal-0/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/contour-internal-0/custom-values.yaml deleted file mode 100644 index a0a66f6..0000000 --- a/helm-overrides/k8s-central-prd-ase1/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,104 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-0-arm - nodeSelector: - cloud.google.com/compute-class: contour-internal-0-arm - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-central-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-central-prd-ase1/contour-internal-1/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/contour-internal-1/custom-values.yaml deleted file mode 100644 index 6b4279b..0000000 --- a/helm-overrides/k8s-central-prd-ase1/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,108 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 360s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-1-arm - - effect: NoSchedule - key: kubernetes.io/arch - operator: Equal - value: arm64 - nodeSelector: - cloud.google.com/compute-class: contour-internal-1-arm - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int1-central-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-central-prd-ase1/contour-internal-intra-0/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index 69c502d..0000000 --- a/helm-overrides/k8s-central-prd-ase1/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,101 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-intra-0-arm - nodeSelector: - cloud.google.com/compute-class: contour-intra-0-arm - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-central-prd-ase1/contour-internal-intra-1/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index f7e4568..0000000 --- a/helm-overrides/k8s-central-prd-ase1/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,102 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 360s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-intra-1-arm - nodeSelector: - cloud.google.com/compute-class: contour-intra-1-arm - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-central-prd-ase1/coredns/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/coredns/custom-values.yaml deleted file mode 100644 index 5d7fd62..0000000 --- a/helm-overrides/k8s-central-prd-ase1/coredns/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -replicaCount: 24 -communicationType: "intra" - -labels: - bu: central - team: central-devops - env: prd - -clusterIP: 10.137.36.2 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - -nodeSelector: - dedicated: central-devops - kubernetes.io/os: linux diff --git a/helm-overrides/k8s-central-prd-ase1/coroot-node-agent/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/coroot-node-agent/custom-values.yaml deleted file mode 100644 index 06bbb8f..0000000 --- a/helm-overrides/k8s-central-prd-ase1/coroot-node-agent/custom-values.yaml +++ /dev/null @@ -1,38 +0,0 @@ - -fullnameOverride: "coroot-node-agent-central-prd" -priorityClassName: "low-priority" - -scrape: "false" # disable scraping for now - -resources: - requests: - cpu: "300m" - memory: "600Mi" - limits: - cpu: "500m" - memory: "600Mi" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmstorage - - megaduo - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external - - key: node_pool - operator: NotIn - values: - - np-central-default-prd-ase1 \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1/elasticsearch-mcp/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/elasticsearch-mcp/custom-values.yaml deleted file mode 100644 index 1e0e8d2..0000000 --- a/helm-overrides/k8s-central-prd-ase1/elasticsearch-mcp/custom-values.yaml +++ /dev/null @@ -1,95 +0,0 @@ -fullnameOverride: "elasticsearch-mcp" - -replicas: 1 - -image: - repository: docker.elastic.co/mcp/elasticsearch - tag: "0.4.6" - pullPolicy: IfNotPresent - -labels: - bu: central - team: devops - service: elasticsearch-mcp - env: prd - -serviceAccount: - create: true - annotations: {} - -podAnnotations: {} - -podSecurityContext: {} - -securityContext: - runAsNonRoot: true - runAsUser: 1000 - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -priorityClassName: "" - -nodeSelector: - dedicated: "devops-mcp" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops-mcp" - effect: NoSchedule - -affinity: {} - -resources: - requests: - cpu: 500m - memory: 1Gi - limits: - cpu: 2000m - memory: 2Gi - -livenessProbe: - httpGet: - path: / - port: 8080 - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - -readinessProbe: - httpGet: - path: / - port: 8080 - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - -externalSecrets: - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/prd/cntr/devop/elasticsearch-mcp" - -createContourGateway: true - -ingress: - enabled: true - ingressClassName: contour-internal-1 - servicePortNumber: 8080 - hosts: - - host: elastic-mcp.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 diff --git a/helm-overrides/k8s-central-prd-ase1/etcd/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/etcd/custom-values.yaml deleted file mode 100644 index de366c3..0000000 --- a/helm-overrides/k8s-central-prd-ase1/etcd/custom-values.yaml +++ /dev/null @@ -1,1105 +0,0 @@ -# Copyright Broadcom, Inc. All Rights Reserved. -# SPDX-License-Identifier: APACHE-2.0 - -## @section Global parameters -## Global Docker image parameters -## Please, note that this will override the image parameters, including dependencies, configured to use the global value -## Current available global Docker image parameters: imageRegistry, imagePullSecrets and storageClass -## - -## @param global.imageRegistry Global Docker image registry -## @param global.imagePullSecrets [array] Global Docker registry secret names as an array -## @param global.defaultStorageClass Global default StorageClass for Persistent Volume(s) -## @param global.storageClass DEPRECATED: use global.defaultStorageClass instead -## -global: - imageRegistry: "" - ## E.g. - ## imagePullSecrets: - ## - myRegistryKeySecretName - ## - imagePullSecrets: [] - defaultStorageClass: "" - storageClass: "" - ## Compatibility adaptations for Kubernetes platforms - ## - compatibility: - ## Compatibility adaptations for Openshift - ## - openshift: - ## @param global.compatibility.openshift.adaptSecurityContext Adapt the securityContext sections of the deployment to make them compatible with Openshift restricted-v2 SCC: remove runAsUser, runAsGroup and fsGroup and let the platform use their allowed default IDs. Possible values: auto (apply if the detected running cluster is Openshift), force (perform the adaptation always), disabled (do not perform adaptation) - ## - adaptSecurityContext: auto -## @section Common parameters -## - -## @param kubeVersion Force target Kubernetes version (using Helm capabilities if not set) -## -kubeVersion: "" -## @param nameOverride String to partially override common.names.fullname template (will maintain the release name) -## -nameOverride: "" -## @param fullnameOverride String to fully override common.names.fullname template -## -fullnameOverride: "" -## @param commonLabels [object] Labels to add to all deployed objects -## -commonLabels: {} -## @param commonAnnotations [object] Annotations to add to all deployed objects -## -commonAnnotations: {} -## @param clusterDomain Default Kubernetes cluster domain -## -clusterDomain: cluster.local -## @param extraDeploy [array] Array of extra objects to deploy with the release -## -extraDeploy: [] -## Enable diagnostic mode in the deployment -## -diagnosticMode: - ## @param diagnosticMode.enabled Enable diagnostic mode (all probes will be disabled and the command will be overridden) - ## - enabled: false - ## @param diagnosticMode.command Command to override all containers in the deployment - ## - command: - - sleep - ## @param diagnosticMode.args Args to override all containers in the deployment - ## - args: - - infinity -## @section etcd parameters -## - -## Bitnami etcd image version -## ref: https://hub.docker.com/r/bitnami/etcd/tags/ -## @param image.registry [default: REGISTRY_NAME] etcd image registry -## @param image.repository [default: REPOSITORY_NAME/etcd] etcd image name -## @skip image.tag etcd image tag -## @param image.digest etcd image digest in the way sha256:aa.... Please note this parameter, if set, will override the tag -## -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/devops/bitnami/etcd - tag: 3.5.16-debian-12-r2 - digest: "" - ## @param image.pullPolicy etcd image pull policy - ## Specify a imagePullPolicy - ## Defaults to 'Always' if image tag is 'latest', else set to 'IfNotPresent' - ## ref: https://kubernetes.io/docs/concepts/containers/images/#pre-pulled-images - ## - pullPolicy: IfNotPresent - ## @param image.pullSecrets [array] etcd image pull secrets - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## e.g: - ## pullSecrets: - ## - myRegistryKeySecretName - ## - pullSecrets: [] - ## @param image.debug Enable image debug mode - ## Set to true if you would like to see extra information on logs - ## - debug: false -## Authentication parameters -## -auth: - ## Role-based access control parameters - ## ref: https://etcd.io/docs/current/op-guide/authentication/ - ## - rbac: - ## @param auth.rbac.create Switch to enable RBAC authentication - ## - create: true - ## @param auth.rbac.allowNoneAuthentication Allow to use etcd without configuring RBAC authentication - ## - allowNoneAuthentication: true - ## @param auth.rbac.rootPassword Root user password. The root user is always `root` - ## - rootPassword: "" - ## @param auth.rbac.existingSecret Name of the existing secret containing credentials for the root user - ## - existingSecret: "etcd-root-password" - ## @param auth.rbac.existingSecretPasswordKey Name of key containing password to be retrieved from the existing secret - ## - existingSecretPasswordKey: "rootPassword" - ## Authentication token - ## ref: https://etcd.io/docs/latest/learning/design-auth-v3/#two-types-of-tokens-simple-and-jwt - ## - token: - ## @param auth.token.enabled Enables token authentication - ## - enabled: true - ## @param auth.token.type Authentication token type. Allowed values: 'simple' or 'jwt' - ## ref: https://etcd.io/docs/latest/op-guide/configuration/#--auth-token - ## - type: jwt - ## @param auth.token.privateKey.filename Name of the file containing the private key for signing the JWT token - ## @param auth.token.privateKey.existingSecret Name of the existing secret containing the private key for signing the JWT token - ## NOTE: Ignored if auth.token.type=simple - ## NOTE: A secret containing a private key will be auto-generated if an existing one is not provided. - ## - privateKey: - filename: jwt-token.pem - existingSecret: "" - ## @param auth.token.signMethod JWT token sign method - ## NOTE: Ignored if auth.token.type=simple - ## - signMethod: RS256 - ## @param auth.token.ttl JWT token TTL - ## NOTE: Ignored if auth.token.type=simple - ## - ttl: 10m - ## TLS authentication for client-to-server communications - ## ref: https://etcd.io/docs/current/op-guide/security/ - ## - client: - ## @param auth.client.secureTransport Switch to encrypt client-to-server communications using TLS certificates - ## - secureTransport: false - ## @param auth.client.useAutoTLS Switch to automatically create the TLS certificates - ## - useAutoTLS: false - ## @param auth.client.existingSecret Name of the existing secret containing the TLS certificates for client-to-server communications - ## - existingSecret: "" - ## @param auth.client.enableAuthentication Switch to enable host authentication using TLS certificates. Requires existing secret - ## - enableAuthentication: false - ## @param auth.client.certFilename Name of the file containing the client certificate - ## - certFilename: cert.pem - ## @param auth.client.certKeyFilename Name of the file containing the client certificate private key - ## - certKeyFilename: key.pem - ## @param auth.client.caFilename Name of the file containing the client CA certificate - ## If not specified and `auth.client.enableAuthentication=true` or `auth.rbac.enabled=true`, the default is is `ca.crt` - ## - caFilename: "" - ## TLS authentication for server-to-server communications - ## ref: https://etcd.io/docs/current/op-guide/security/ - ## - peer: - ## @param auth.peer.secureTransport Switch to encrypt server-to-server communications using TLS certificates - ## - secureTransport: false - ## @param auth.peer.useAutoTLS Switch to automatically create the TLS certificates - ## - useAutoTLS: false - ## @param auth.peer.existingSecret Name of the existing secret containing the TLS certificates for server-to-server communications - ## - existingSecret: "" - ## @param auth.peer.enableAuthentication Switch to enable host authentication using TLS certificates. Requires existing secret - ## - enableAuthentication: false - ## @param auth.peer.certFilename Name of the file containing the peer certificate - ## - certFilename: cert.pem - ## @param auth.peer.certKeyFilename Name of the file containing the peer certificate private key - ## - certKeyFilename: key.pem - ## @param auth.peer.caFilename Name of the file containing the peer CA certificate - ## If not specified and `auth.peer.enableAuthentication=true` or `rbac.enabled=true`, the default is is `ca.crt` - ## - caFilename: "" -## @param autoCompactionMode Auto compaction mode, by default periodic. Valid values: "periodic", "revision". -## - 'periodic' for duration based retention, defaulting to hours if no time unit is provided (e.g. 5m). -## - 'revision' for revision number based retention. -## -autoCompactionMode: "" -## @param autoCompactionRetention Auto compaction retention for mvcc key value store in hour, by default 0, means disabled -## -autoCompactionRetention: "" -## @param initialClusterState Initial cluster state. Allowed values: 'new' or 'existing' -## If this values is not set, the default values below are set: -## - 'new': when installing the chart ('helm install ...') -## - 'existing': when upgrading the chart ('helm upgrade ...') -## -initialClusterState: "" -## @param initialClusterToken Initial cluster token. Can be used to protect etcd from cross-cluster-interaction, which might corrupt the clusters. -## If spinning up multiple clusters (or creating and destroying a single cluster) -## with same configuration for testing purpose, it is highly recommended that each cluster is given a unique initial-cluster-token. -## By doing this, etcd can generate unique cluster IDs and member IDs for the clusters even if they otherwise have the exact same configuration. -## -initialClusterToken: "etcd-cluster-k8s" -## @param logLevel Sets the log level for the etcd process. Allowed values: 'debug', 'info', 'warn', 'error', 'panic', 'fatal' -## -logLevel: "info" -## @param maxProcs Limits the number of operating system threads that can execute user-level -## Go code simultaneously by setting GOMAXPROCS environment variable -## ref: https://golang.org/pkg/runtime -## -maxProcs: "" -## @param removeMemberOnContainerTermination Use a PreStop hook to remove the etcd members from the etcd cluster on container termination -## they the containers are terminated. Set to 'false' if appears an error-related member ID wasn't properly stored. -## NOTE: Ignored if lifecycleHooks is set or replicaCount=1 -## -removeMemberOnContainerTermination: true -## @param configuration etcd configuration. Specify content for etcd.conf.yml -## e.g: -## configuration: |- -## foo: bar -## baz: -## -configuration: "" -## @param existingConfigmap Existing ConfigMap with etcd configuration -## NOTE: When it's set the configuration parameter is ignored -## -existingConfigmap: "" -## @param extraEnvVars [array] Extra environment variables to be set on etcd container -## e.g: -## extraEnvVars: -## - name: FOO -## value: "bar" -## -extraEnvVars: [] -## @param extraEnvVarsCM Name of existing ConfigMap containing extra env vars -## -extraEnvVarsCM: "" -## @param extraEnvVarsSecret Name of existing Secret containing extra env vars -## -extraEnvVarsSecret: "" -## @param command [array] Default container command (useful when using custom images) -## -command: [] -## @param args [array] Default container args (useful when using custom images) -## -args: [] -## @section etcd statefulset parameters -## - -## @param replicaCount Number of etcd replicas to deploy -## -replicaCount: 1 -## Update strategy -## ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies -## @param updateStrategy.type Update strategy type, can be set to RollingUpdate or OnDelete. -## -updateStrategy: - type: RollingUpdate -## @param podManagementPolicy Pod management policy for the etcd statefulset -## ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#pod-management-policies -## -podManagementPolicy: Parallel -## @param automountServiceAccountToken Mount Service Account token in pod -## -automountServiceAccountToken: false -## @param hostAliases [array] etcd pod host aliases -## ref: https://kubernetes.io/docs/concepts/services-networking/add-entries-to-pod-etc-hosts-with-host-aliases/ -## -hostAliases: [] -## @param lifecycleHooks [object] Override default etcd container hooks -## -lifecycleHooks: {} -## etcd container ports to open -## @param containerPorts.client Client port to expose at container level -## @param containerPorts.peer Peer port to expose at container level -## @param containerPorts.metrics Metrics port to expose at container level when metrics.useSeparateEndpoint is true -## -containerPorts: - client: 2379 - peer: 2380 - metrics: 9090 -## etcd pods' Security Context -## ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-pod -## @param podSecurityContext.enabled Enabled etcd pods' Security Context -## @param podSecurityContext.fsGroupChangePolicy Set filesystem group change policy -## @param podSecurityContext.sysctls Set kernel settings using the sysctl interface -## @param podSecurityContext.supplementalGroups Set filesystem extra groups -## @param podSecurityContext.fsGroup Set etcd pod's Security Context fsGroup -## -podSecurityContext: - enabled: true - fsGroupChangePolicy: Always - sysctls: [] - supplementalGroups: [] - fsGroup: 1001 -## etcd containers' SecurityContext -## ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -## @param containerSecurityContext.enabled Enabled etcd containers' Security Context -## @param containerSecurityContext.seLinuxOptions [object,nullable] Set SELinux options in container -## @param containerSecurityContext.runAsUser Set etcd containers' Security Context runAsUser -## @param containerSecurityContext.runAsGroup Set etcd containers' Security Context runAsUser -## @param containerSecurityContext.runAsNonRoot Set Controller container's Security Context runAsNonRoot -## @param containerSecurityContext.privileged Set primary container's Security Context privileged -## @param containerSecurityContext.allowPrivilegeEscalation Set primary container's Security Context allowPrivilegeEscalation -## @param containerSecurityContext.readOnlyRootFilesystem Set container's Security Context readOnlyRootFilesystem -## @param containerSecurityContext.capabilities.drop List of capabilities to be dropped -## @param containerSecurityContext.seccompProfile.type Set container's Security Context seccomp profile -## -containerSecurityContext: - enabled: true - seLinuxOptions: {} - runAsUser: 1001 - runAsGroup: 1001 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: ["ALL"] - seccompProfile: - type: "RuntimeDefault" -## etcd containers' resource requests and limits -## ref: https://kubernetes.io/docs/concepts/configuration/manage-compute-resources-container/ -## We usually recommend not to specify default resources and to leave this as a conscious -## choice for the user. This also increases chances charts run on environments with little -## resources, such as Minikube. If you do want to specify resources, uncomment the following -## lines, adjust them as necessary, and remove the curly braces after 'resources:'. -## @param resourcesPreset Set container resources according to one common preset (allowed values: none, nano, micro, small, medium, large, xlarge, 2xlarge). This is ignored if resources is set (resources is recommended for production). -## More information: https://github.com/bitnami/charts/blob/main/bitnami/common/templates/_resources.tpl#L15 -## -resourcesPreset: "micro" -## @param resources Set container requests and limits for different resources like CPU or memory (essential for production workloads) -## Example: -## resources: -## requests: -## cpu: 2 -## memory: 512Mi -## limits: -## cpu: 3 -## memory: 1024Mi -## -resources: {} -## Configure extra options for liveness probe -## ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/#configure-probes -## @param livenessProbe.enabled Enable livenessProbe -## @param livenessProbe.initialDelaySeconds Initial delay seconds for livenessProbe -## @param livenessProbe.periodSeconds Period seconds for livenessProbe -## @param livenessProbe.timeoutSeconds Timeout seconds for livenessProbe -## @param livenessProbe.failureThreshold Failure threshold for livenessProbe -## @param livenessProbe.successThreshold Success threshold for livenessProbe -## -livenessProbe: - enabled: true - initialDelaySeconds: 60 - periodSeconds: 30 - timeoutSeconds: 5 - successThreshold: 1 - failureThreshold: 5 -## Configure extra options for readiness probe -## ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/#configure-probes -## @param readinessProbe.enabled Enable readinessProbe -## @param readinessProbe.initialDelaySeconds Initial delay seconds for readinessProbe -## @param readinessProbe.periodSeconds Period seconds for readinessProbe -## @param readinessProbe.timeoutSeconds Timeout seconds for readinessProbe -## @param readinessProbe.failureThreshold Failure threshold for readinessProbe -## @param readinessProbe.successThreshold Success threshold for readinessProbe -## -readinessProbe: - enabled: true - initialDelaySeconds: 60 - periodSeconds: 10 - timeoutSeconds: 5 - successThreshold: 1 - failureThreshold: 5 -## Configure extra options for liveness probe -## ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/#configure-probes -## @param startupProbe.enabled Enable startupProbe -## @param startupProbe.initialDelaySeconds Initial delay seconds for startupProbe -## @param startupProbe.periodSeconds Period seconds for startupProbe -## @param startupProbe.timeoutSeconds Timeout seconds for startupProbe -## @param startupProbe.failureThreshold Failure threshold for startupProbe -## @param startupProbe.successThreshold Success threshold for startupProbe -## -startupProbe: - enabled: false - initialDelaySeconds: 0 - periodSeconds: 10 - timeoutSeconds: 5 - successThreshold: 1 - failureThreshold: 60 -## @param customLivenessProbe [object] Override default liveness probe -## -customLivenessProbe: {} -## @param customReadinessProbe [object] Override default readiness probe -## -customReadinessProbe: {} -## @param customStartupProbe [object] Override default startup probe -## -customStartupProbe: {} -## @param extraVolumes [array] Optionally specify extra list of additional volumes for etcd pods -## -extraVolumes: [] -## @param extraVolumeMounts [array] Optionally specify extra list of additional volumeMounts for etcd container(s) -## -extraVolumeMounts: [] -## @param extraVolumeClaimTemplates [array] Optionally specify extra list of additional volumeClaimTemplates for etcd container(s) -## -extraVolumeClaimTemplates: [] -## @param initContainers [array] Add additional init containers to the etcd pods -## e.g: -## initContainers: -## - name: your-image-name -## image: your-image -## imagePullPolicy: Always -## ports: -## - name: portname -## containerPort: 1234 -## -initContainers: [] -## @param sidecars [array] Add additional sidecar containers to the etcd pods -## e.g: -## sidecars: -## - name: your-image-name -## image: your-image -## imagePullPolicy: Always -## ports: -## - name: portname -## containerPort: 1234 -## -sidecars: [] -## @param podAnnotations [object] Annotations for etcd pods -## ref: https://kubernetes.io/docs/concepts/overview/working-with-objects/annotations/ -## -podAnnotations: {} -## @param podLabels [object] Extra labels for etcd pods -## Ref: https://kubernetes.io/docs/concepts/overview/working-with-objects/labels/ -## -podLabels: {} -## @param podAffinityPreset Pod affinity preset. Ignored if `affinity` is set. Allowed values: `soft` or `hard` -## ref: https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#inter-pod-affinity-and-anti-affinity -## -podAffinityPreset: "" -## @param podAntiAffinityPreset Pod anti-affinity preset. Ignored if `affinity` is set. Allowed values: `soft` or `hard` -## Ref: https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#inter-pod-affinity-and-anti-affinity -## -podAntiAffinityPreset: soft -## Node affinity preset -## Ref: https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#node-affinity -## @param nodeAffinityPreset.type Node affinity preset type. Ignored if `affinity` is set. Allowed values: `soft` or `hard` -## @param nodeAffinityPreset.key Node label key to match. Ignored if `affinity` is set. -## @param nodeAffinityPreset.values [array] Node label values to match. Ignored if `affinity` is set. -## -nodeAffinityPreset: - type: "" - ## e.g: - ## key: "kubernetes.io/e2e-az-name" - ## - key: "" - ## e.g: - ## values: - ## - e2e-az1 - ## - e2e-az2 - ## - values: [] -## @param affinity [object] Affinity for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## Note: podAffinityPreset, podAntiAffinityPreset, and nodeAffinityPreset will be ignored when it's set -## -affinity: {} -## @param nodeSelector [object] Node labels for pod assignment -## Ref: https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/ -## -nodeSelector: {} -## @param tolerations [array] Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -## -tolerations: [] -## @param terminationGracePeriodSeconds Seconds the pod needs to gracefully terminate -## ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/#hook-handler-execution -## -terminationGracePeriodSeconds: "" -## @param schedulerName Name of the k8s scheduler (other than default) -## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ -## -schedulerName: "" -## @param priorityClassName Name of the priority class to be used by etcd pods -## Priority class needs to be created beforehand -## Ref: https://kubernetes.io/docs/concepts/configuration/pod-priority-preemption/ -## -priorityClassName: "" -## @param runtimeClassName Name of the runtime class to be used by pod(s) -## ref: https://kubernetes.io/docs/concepts/containers/runtime-class/ -## -runtimeClassName: "" -## @param shareProcessNamespace Enable shared process namespace in a pod. -## If set to false (default), each container will run in separate namespace, etcd will have PID=1. -## If set to true, the /pause will run as init process and will reap any zombie PIDs, -## for example, generated by a custom exec probe running longer than a probe timeoutSeconds. -## Enable this only if customLivenessProbe or customReadinessProbe is used and zombie PIDs are accumulating. -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/share-process-namespace/ -## -shareProcessNamespace: false -## @param topologySpreadConstraints Topology Spread Constraints for pod assignment -## https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -## The value is evaluated as a template -## -topologySpreadConstraints: [] -## persistentVolumeClaimRetentionPolicy -## ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#persistentvolumeclaim-retention -## @param persistentVolumeClaimRetentionPolicy.enabled Controls if and how PVCs are deleted during the lifecycle of a StatefulSet -## @param persistentVolumeClaimRetentionPolicy.whenScaled Volume retention behavior when the replica count of the StatefulSet is reduced -## @param persistentVolumeClaimRetentionPolicy.whenDeleted Volume retention behavior that applies when the StatefulSet is deleted -persistentVolumeClaimRetentionPolicy: - enabled: false - whenScaled: Retain - whenDeleted: Retain -## @section Traffic exposure parameters -## - -service: - ## @param service.type Kubernetes Service type - ## - type: ClusterIP - ## @param service.enabled create second service if equal true - ## - enabled: true - ## @param service.clusterIP Kubernetes service Cluster IP - ## e.g.: - ## clusterIP: None - ## - clusterIP: "" - ## @param service.ports.client etcd client port - ## @param service.ports.peer etcd peer port - ## @param service.ports.metrics etcd metrics port when metrics.useSeparateEndpoint is true - ## - ports: - client: 2379 - peer: 2380 - metrics: 9090 - ## @param service.nodePorts.client Specify the nodePort client value for the LoadBalancer and NodePort service types. - ## @param service.nodePorts.peer Specify the nodePort peer value for the LoadBalancer and NodePort service types. - ## @param service.nodePorts.metrics Specify the nodePort metrics value for the LoadBalancer and NodePort service types. The metrics port is only exposed when metrics.useSeparateEndpoint is true. - ## ref: https://kubernetes.io/docs/concepts/services-networking/service/#type-nodeport - ## - nodePorts: - client: "" - peer: "" - metrics: "" - ## @param service.clientPortNameOverride etcd client port name override - ## - clientPortNameOverride: "" - ## @param service.peerPortNameOverride etcd peer port name override - ## - peerPortNameOverride: "" - ## @param service.metricsPortNameOverride etcd metrics port name override. The metrics port is only exposed when metrics.useSeparateEndpoint is true. - ## - metricsPortNameOverride: "" - ## @param service.loadBalancerIP loadBalancerIP for the etcd service (optional, cloud specific) - ## ref: https://kubernetes.io/docs/concepts/services-networking/service/#type-loadbalancer - ## - loadBalancerIP: "" - ## @param service.loadBalancerSourceRanges [array] Load Balancer source ranges - ## ref: https://kubernetes.io/docs/tasks/access-application-cluster/configure-cloud-provider-firewall/#restrict-access-for-loadbalancer-service - ## e.g: - ## loadBalancerSourceRanges: - ## - 10.10.10.0/24 - ## - loadBalancerSourceRanges: [] - ## @param service.externalIPs [array] External IPs - ## ref: https://kubernetes.io/docs/concepts/services-networking/service/#external-ips - ## - externalIPs: [] - ## @param service.externalTrafficPolicy %%MAIN_CONTAINER_NAME%% service external traffic policy - ## ref http://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - ## - externalTrafficPolicy: Cluster - ## @param service.extraPorts Extra ports to expose (normally used with the `sidecar` value) - ## - extraPorts: [] - ## @param service.annotations [object] Additional annotations for the etcd service - ## - annotations: {} - ## @param service.sessionAffinity Session Affinity for Kubernetes service, can be "None" or "ClientIP" - ## If "ClientIP", consecutive client requests will be directed to the same Pod - ## ref: https://kubernetes.io/docs/concepts/services-networking/service/#virtual-ips-and-service-proxies - ## - sessionAffinity: None - ## @param service.sessionAffinityConfig Additional settings for the sessionAffinity - ## sessionAffinityConfig: - ## clientIP: - ## timeoutSeconds: 300 - ## - sessionAffinityConfig: {} - ## Headless service properties - ## - headless: - ## @param service.headless.annotations Annotations for the headless service. - ## - annotations: {} -## @section Persistence parameters -## - -## Enable persistence using Persistent Volume Claims -## ref: https://kubernetes.io/docs/concepts/storage/persistent-volumes/ -## -persistence: - ## @param persistence.enabled If true, use a Persistent Volume Claim. If false, use emptyDir. - ## - enabled: true - ## @param persistence.storageClass Persistent Volume Storage Class - ## If defined, storageClassName: - ## If set to "-", storageClassName: "", which disables dynamic provisioning - ## If undefined (the default) or set to null, no storageClassName spec is - ## set, choosing the default provisioner. (gp2 on AWS, standard on - ## GKE, AWS & OpenStack) - ## - storageClass: "sc-pd-standard" - ## - ## @param persistence.annotations [object] Annotations for the PVC - ## - annotations: {} - ## @param persistence.labels [object] Labels for the PVC - ## - labels: {} - ## @param persistence.accessModes Persistent Volume Access Modes - ## - accessModes: - - ReadWriteOnce - ## @param persistence.size PVC Storage Request for etcd data volume - ## - size: 8Gi - ## @param persistence.selector [object] Selector to match an existing Persistent Volume - ## ref: https://kubernetes.io/docs/concepts/storage/persistent-volumes/#selector - ## - selector: {} -## @section Volume Permissions parameters -## - -## Init containers parameters: -## volumePermissions: Change the owner and group of the persistent volume mountpoint to runAsUser:fsGroup values from the securityContext section. -## -volumePermissions: - ## @param volumePermissions.enabled Enable init container that changes the owner and group of the persistent volume(s) mountpoint to `runAsUser:fsGroup` - ## - enabled: false - ## @param volumePermissions.image.registry [default: REGISTRY_NAME] Init container volume-permissions image registry - ## @param volumePermissions.image.repository [default: REPOSITORY_NAME/os-shell] Init container volume-permissions image name - ## @skip volumePermissions.image.tag Init container volume-permissions image tag - ## @param volumePermissions.image.digest Init container volume-permissions image digest in the way sha256:aa.... Please note this parameter, if set, will override the tag - ## - image: - registry: docker.io - repository: bitnami/os-shell - tag: 12-debian-12-r30 - digest: "" - ## @param volumePermissions.image.pullPolicy Init container volume-permissions image pull policy - ## - pullPolicy: IfNotPresent - ## @param volumePermissions.image.pullSecrets [array] Specify docker-registry secret names as an array - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## e.g: - ## pullSecrets: - ## - myRegistryKeySecretName - ## - pullSecrets: [] - ## Init container' resource requests and limits - ## ref: https://kubernetes.io/docs/concepts/configuration/manage-compute-resources-container/ - ## We usually recommend not to specify default resources and to leave this as a conscious - ## choice for the user. This also increases chances charts run on environments with little - ## resources, such as Minikube. If you do want to specify resources, uncomment the following - ## lines, adjust them as necessary, and remove the curly braces after 'resources:'. - ## @param volumePermissions.resourcesPreset Set container resources according to one common preset (allowed values: none, nano, micro, small, medium, large, xlarge, 2xlarge). This is ignored if volumePermissions.resources is set (volumePermissions.resources is recommended for production). - ## More information: https://github.com/bitnami/charts/blob/main/bitnami/common/templates/_resources.tpl#L15 - ## - resourcesPreset: "nano" - ## @param volumePermissions.resources Set container requests and limits for different resources like CPU or memory (essential for production workloads) - ## Example: - ## resources: - ## requests: - ## cpu: 2 - ## memory: 512Mi - ## limits: - ## cpu: 3 - ## memory: 1024Mi - ## - resources: {} -## @section Network Policy parameters -## ref: https://kubernetes.io/docs/concepts/services-networking/network-policies/ -## -networkPolicy: - ## @param networkPolicy.enabled Enable creation of NetworkPolicy resources - ## - enabled: true - ## @param networkPolicy.allowExternal Don't require client label for connections - ## When set to false, only pods with the correct client label will have network access to the ports - ## etcd is listening on. When true, etcd will accept connections from any source - ## (with the correct destination port). - ## - allowExternal: true - ## @param networkPolicy.allowExternalEgress Allow the pod to access any range of port and all destinations. - ## - allowExternalEgress: true - ## @param networkPolicy.extraIngress [array] Add extra ingress rules to the NetworkPolicy - ## e.g: - ## extraIngress: - ## - ports: - ## - port: 1234 - ## from: - ## - podSelector: - ## - matchLabels: - ## - role: frontend - ## - podSelector: - ## - matchExpressions: - ## - key: role - ## operator: In - ## values: - ## - frontend - ## - extraIngress: [] - ## @param networkPolicy.extraEgress [array] Add extra ingress rules to the NetworkPolicy - ## e.g: - ## extraEgress: - ## - ports: - ## - port: 1234 - ## to: - ## - podSelector: - ## - matchLabels: - ## - role: frontend - ## - podSelector: - ## - matchExpressions: - ## - key: role - ## operator: In - ## values: - ## - frontend - ## - extraEgress: [] - ## @param networkPolicy.ingressNSMatchLabels [object] Labels to match to allow traffic from other namespaces - ## @param networkPolicy.ingressNSPodMatchLabels [object] Pod labels to match to allow traffic from other namespaces - ## - ingressNSMatchLabels: {} - ingressNSPodMatchLabels: {} -## @section Metrics parameters -## -metrics: - ## @param metrics.enabled Expose etcd metrics - ## - enabled: false - ## @param metrics.useSeparateEndpoint Use a separate endpoint for exposing metrics - # - useSeparateEndpoint: false - ## @param metrics.podAnnotations [object] Annotations for the Prometheus metrics on etcd pods - ## - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "{{ .Values.metrics.useSeparateEndpoint | ternary .Values.containerPorts.metrics .Values.containerPorts.client }}" - ## Prometheus Service Monitor - ## ref: https://github.com/coreos/prometheus-operator/blob/master/Documentation/api.md#endpoint - ## - podMonitor: - ## @param metrics.podMonitor.enabled Create PodMonitor Resource for scraping metrics using PrometheusOperator - ## - enabled: false - ## @param metrics.podMonitor.namespace Namespace in which Prometheus is running - ## - namespace: monitoring - ## @param metrics.podMonitor.interval Specify the interval at which metrics should be scraped - ## - interval: 30s - ## @param metrics.podMonitor.scrapeTimeout Specify the timeout after which the scrape is ended - ## - scrapeTimeout: 30s - ## @param metrics.podMonitor.additionalLabels [object] Additional labels that can be used so PodMonitors will be discovered by Prometheus - ## ref: https://github.com/coreos/prometheus-operator/blob/master/Documentation/api.md#prometheusspec - ## - additionalLabels: {} - ## @param metrics.podMonitor.scheme Scheme to use for scraping - ## - scheme: http - ## @param metrics.podMonitor.tlsConfig [object] TLS configuration used for scrape endpoints used by Prometheus - ## ref: https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#tlsconfig - ## e.g: - ## tlsConfig: - ## ca: - ## secret: - ## name: existingSecretName - ## - tlsConfig: {} - ## @param metrics.podMonitor.relabelings [array] Prometheus relabeling rules - ## - relabelings: [] - ## Prometheus Operator PrometheusRule configuration - ## - prometheusRule: - ## @param metrics.prometheusRule.enabled Create a Prometheus Operator PrometheusRule (also requires `metrics.enabled` to be `true` and `metrics.prometheusRule.rules`) - ## - enabled: false - ## @param metrics.prometheusRule.namespace Namespace for the PrometheusRule Resource (defaults to the Release Namespace) - ## - namespace: "" - ## @param metrics.prometheusRule.additionalLabels Additional labels that can be used so PrometheusRule will be discovered by Prometheus - ## - additionalLabels: {} - ## @param metrics.prometheusRule.rules Prometheus Rule definitions - # - alert: ETCD has no leader - # annotations: - # summary: "ETCD has no leader" - # description: "pod {{`{{`}} $labels.pod {{`}}`}} state error, can't connect leader" - # for: 1m - # expr: etcd_server_has_leader == 0 - # labels: - # severity: critical - # group: PaaS - ## - rules: [] -## @section Snapshotting parameters -## - -## Start a new etcd cluster recovering the data from an existing snapshot before bootstrapping -## -startFromSnapshot: - ## @param startFromSnapshot.enabled Initialize new cluster recovering an existing snapshot - ## - enabled: false - ## @param startFromSnapshot.existingClaim Existing PVC containing the etcd snapshot - ## - existingClaim: "" - ## @param startFromSnapshot.snapshotFilename Snapshot filename - ## - snapshotFilename: "" -## Enable auto disaster recovery by periodically snapshotting the keyspace: -## - It creates a cronjob to periodically snapshotting the keyspace -## - It also creates a ReadWriteMany PVC to store the snapshots -## If the cluster permanently loses more than (N-1)/2 members, it tries to -## recover itself from the last available snapshot. -## -disasterRecovery: - ## @param disasterRecovery.enabled Enable auto disaster recovery by periodically snapshotting the keyspace - ## - enabled: false - cronjob: - ## @param disasterRecovery.cronjob.schedule Schedule in Cron format to save snapshots - ## See https://en.wikipedia.org/wiki/Cron - ## - schedule: "*/30 * * * *" - ## @param disasterRecovery.cronjob.historyLimit Number of successful finished jobs to retain - ## - historyLimit: 1 - ## @param disasterRecovery.cronjob.snapshotHistoryLimit Number of etcd snapshots to retain, tagged by date - ## - snapshotHistoryLimit: 1 - ## @param disasterRecovery.cronjob.snapshotsDir Directory to store snapshots - ## - snapshotsDir: "/snapshots" - ## @param disasterRecovery.cronjob.podAnnotations [object] Pod annotations for cronjob pods - ## ref: https://kubernetes.io/docs/concepts/overview/working-with-objects/annotations/ - ## - podAnnotations: {} - ## Configure resource requests and limits for snapshotter containers - ## ref: https://kubernetes.io/docs/concepts/configuration/manage-compute-resources-container/ - ## We usually recommend not to specify default resources and to leave this as a conscious - ## choice for the user. This also increases chances charts run on environments with little - ## resources, such as Minikube. If you do want to specify resources, uncomment the following - ## lines, adjust them as necessary, and remove the curly braces after 'resources:'. - ## @param disasterRecovery.cronjob.resourcesPreset Set container resources according to one common preset (allowed values: none, nano, micro, small, medium, large, xlarge, 2xlarge). This is ignored if disasterRecovery.cronjob.resources is set (disasterRecovery.cronjob.resources is recommended for production). - ## More information: https://github.com/bitnami/charts/blob/main/bitnami/common/templates/_resources.tpl#L15 - ## - resourcesPreset: "nano" - ## @param disasterRecovery.cronjob.resources Set container requests and limits for different resources like CPU or memory (essential for production workloads) - ## Example: - ## resources: - ## requests: - ## cpu: 2 - ## memory: 512Mi - ## limits: - ## cpu: 3 - ## memory: 1024Mi - ## - resources: {} - ## @param disasterRecovery.cronjob.nodeSelector Node labels for cronjob pods assignment - ## Ref: https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/ - ## - nodeSelector: {} - ## @param disasterRecovery.cronjob.tolerations Tolerations for cronjob pods assignment - ## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ - ## - tolerations: [] - ## @param disasterRecovery.cronjob.podLabels [object] Labels that will be added to pods created by cronjob - ## - podLabels: {} - ## @param disasterRecovery.cronjob.serviceAccountName Specifies the service account to use for disaster recovery cronjob - ## - serviceAccountName: "" - ## @param disasterRecovery.cronjob.command Override default snapshot container command (useful when you want to customize the snapshot logic) - ## - command: [] - ## - pvc: - ## @param disasterRecovery.pvc.existingClaim A manually managed Persistent Volume and Claim - ## If defined, PVC must be created manually before volume will be bound - ## The value is evaluated as a template, so, for example, the name can depend on .Release or .Chart - ## - existingClaim: "" - ## @param disasterRecovery.pvc.size PVC Storage Request - ## - size: 2Gi - ## @param disasterRecovery.pvc.storageClassName Storage Class for snapshots volume - ## - storageClassName: nfs - ## @param disasterRecovery.pvc.subPath Path within the volume from which to mount - ## Useful if snapshots should only be stored in a subdirectory of the volume - ## - subPath: "" -## @section Service account parameters -## -serviceAccount: - ## @param serviceAccount.create Enable/disable service account creation - ## - create: true - ## @param serviceAccount.name Name of the service account to create or use - ## - name: "" - ## @param serviceAccount.automountServiceAccountToken Enable/disable auto mounting of service account token - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-service-account/#use-the-default-service-account-to-access-the-api-server - ## - automountServiceAccountToken: false - ## @param serviceAccount.annotations [object] Additional annotations to be included on the service account - ## - annotations: {} - ## @param serviceAccount.labels [object] Additional labels to be included on the service account - ## - labels: {} -## @section Defragmentation parameters -## - -## Enable defragmentation by periodically rearranging fragmented data after history compaction. -## It creates a cronjob to periodically run the defragmentation command: -## etcdctl defrag [OPTIONS] -## See https://etcd.io/docs/latest/op-guide/maintenance/ -## -defrag: - ## @param defrag.enabled Enable automatic defragmentation. This is most effective when paired with auto compaction: consider setting "autoCompactionRetention > 0". - ## - enabled: false - cronjob: - ## @param defrag.cronjob.startingDeadlineSeconds Number of seconds representing the deadline for starting the job if it misses scheduled time for any reason - ## - startingDeadlineSeconds: "" - ## @param defrag.cronjob.schedule Schedule in Cron format to defrag (daily at midnight by default) - ## See https://en.wikipedia.org/wiki/Cron - ## - schedule: "0 0 * * *" - ## @param defrag.cronjob.concurrencyPolicy Set the cronjob parameter concurrencyPolicy - ## - concurrencyPolicy: Forbid - ## @param defrag.cronjob.suspend Boolean that indicates if the controller must suspend subsequent executions (not applied to already started executions) - ## - suspend: false - ## @param defrag.cronjob.successfulJobsHistoryLimit Number of successful finished jobs to retain - ## - successfulJobsHistoryLimit: 1 - ## @param defrag.cronjob.failedJobsHistoryLimit Number of failed finished jobs to retain - ## - failedJobsHistoryLimit: 1 - ## @param defrag.cronjob.labels [object] Additional labels to be added to the Defrag cronjob - ## - labels: {} - ## @param defrag.cronjob.annotations [object] Annotations to be added to the Defrag cronjob - ## - annotations: {} - ## @param defrag.cronjob.activeDeadlineSeconds Number of seconds relative to the startTime that the job may be continuously active before the system tries to terminate it - ## - activeDeadlineSeconds: "" - ## @param defrag.cronjob.restartPolicy Set the cronjob parameter restartPolicy - ## - restartPolicy: OnFailure - ## @param defrag.cronjob.podLabels [object] Labels that will be added to pods created by Defrag cronjob - ## - podLabels: {} - ## @param defrag.cronjob.podAnnotations [object] Pod annotations for Defrag cronjob pods - ## ref: https://kubernetes.io/docs/concepts/overview/working-with-objects/annotations/ - ## - podAnnotations: {} - ## K8s Security Context for Defrag cronjob pods - ## https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - ## @param defrag.cronjob.podSecurityContext.enabled Enable security context for Defrag pods - ## @param defrag.cronjob.podSecurityContext.fsGroupChangePolicy Set filesystem group change policy - ## @param defrag.cronjob.podSecurityContext.sysctls Set kernel settings using the sysctl interface - ## @param defrag.cronjob.podSecurityContext.supplementalGroups Set filesystem extra groups - ## @param defrag.cronjob.podSecurityContext.fsGroup Group ID for the Defrag filesystem - ## - podSecurityContext: - enabled: true - fsGroupChangePolicy: Always - sysctls: [] - supplementalGroups: [] - fsGroup: 1001 - ## Configure container security context for Defrag cronjob pods - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-pod - ## @param defrag.cronjob.containerSecurityContext.enabled Enabled containers' Security Context - ## @param defrag.cronjob.containerSecurityContext.seLinuxOptions [object,nullable] Set SELinux options in container - ## @param defrag.cronjob.containerSecurityContext.runAsUser Set containers' Security Context runAsUser - ## @param defrag.cronjob.containerSecurityContext.runAsGroup Set containers' Security Context runAsGroup - ## @param defrag.cronjob.containerSecurityContext.runAsNonRoot Set container's Security Context runAsNonRoot - ## @param defrag.cronjob.containerSecurityContext.privileged Set container's Security Context privileged - ## @param defrag.cronjob.containerSecurityContext.readOnlyRootFilesystem Set container's Security Context readOnlyRootFilesystem - ## @param defrag.cronjob.containerSecurityContext.allowPrivilegeEscalation Set container's Security Context allowPrivilegeEscalation - ## @param defrag.cronjob.containerSecurityContext.capabilities.drop List of capabilities to be dropped - ## @param defrag.cronjob.containerSecurityContext.seccompProfile.type Set container's Security Context seccomp profile - ## - containerSecurityContext: - enabled: true - seLinuxOptions: {} - runAsUser: 1001 - runAsGroup: 1001 - runAsNonRoot: true - privileged: false - readOnlyRootFilesystem: true - allowPrivilegeEscalation: false - capabilities: - drop: ["ALL"] - seccompProfile: - type: "RuntimeDefault" - ## @param defrag.cronjob.nodeSelector [object] Node labels for pod assignment in Defrag cronjob - ## Ref: https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/ - ## - nodeSelector: {} - ## @param defrag.cronjob.tolerations [array] Tolerations for pod assignment in Defrag cronjob - ## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ - ## - tolerations: [] - ## @param defrag.cronjob.serviceAccountName Specifies the service account to use for Defrag cronjob - ## - serviceAccountName: "" - ## @param defrag.cronjob.command [array] Override default container command for defragmentation (useful when using custom images) - ## ref: https://kubernetes.io/docs/concepts/overview/working-with-objects/annotations/ - ## - command: [] - ## @param defrag.cronjob.args [array] Override default container args (useful when using custom images) - ## - args: [] - ## @param defrag.cronjob.resourcesPreset Set container resources according to one common preset - ## (allowed values: none, nano, micro, small, medium, large, xlarge, 2xlarge). This is ignored if - ## defrag.cronjob.resources is set (defrag.cronjob.resources is recommended for production). - ## More information: https://github.com/bitnami/charts/blob/main/bitnami/common/templates/_resources.tpl#L15 - ## - resourcesPreset: "nano" - ## @param defrag.cronjob.resources [object] Set container requests and limits for different resources like CPU or - ## memory (essential for production workloads) - ## Example: - ## resources: - ## requests: - ## cpu: 2 - ## memory: 512Mi - ## limits: - ## cpu: 3 - ## memory: 1024Mi - ## - resources: {} -## @section Other parameters -## - -## etcd Pod Disruption Budget configuration -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -## -pdb: - ## @param pdb.create Enable/disable a Pod Disruption Budget creation - ## - create: true - ## @param pdb.minAvailable Minimum number/percentage of pods that should remain scheduled - ## - minAvailable: 51% - ## @param pdb.maxUnavailable Maximum number/percentage of pods that may be made unavailable - ## - maxUnavailable: "" - - -#httpproxy values -httpProxy: - enabled: true - ingressClassName: contour-internal-1 - virtualhost: etcd-central.int.meesho.int diff --git a/helm-overrides/k8s-central-prd-ase1/external-secrets/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/external-secrets/custom-values.yaml deleted file mode 100644 index 249b2d0..0000000 --- a/helm-overrides/k8s-central-prd-ase1/external-secrets/custom-values.yaml +++ /dev/null @@ -1,62 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - - # -- Annotations to add to Pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - nodeSelector: - dedicated: central-devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - nodeSelector: - dedicated: central-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - nodeSelector: - dedicated: central-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" diff --git a/helm-overrides/k8s-central-prd-ase1/fireworks-ai/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/fireworks-ai/custom-values.yaml deleted file mode 100644 index 2f8d8f2..0000000 --- a/helm-overrides/k8s-central-prd-ase1/fireworks-ai/custom-values.yaml +++ /dev/null @@ -1,278 +0,0 @@ -# Central production values for the Bifrost 1.5.12 upgrade candidate. -# Chart: helm-templates/bifrost-v1.5.12 - -replicaCount: 1 - -fullnameOverride: "fireworks-ai" - -image: - repository: docker.io/maximhq/bifrost - pullPolicy: IfNotPresent - tag: "v1.5.12" - -serviceAccount: - create: true - automount: true - annotations: - iam.gke.io/gcp-service-account: sa-dvops-ai-gateway@meesho-central-prd-0622.iam.gserviceaccount.com - name: "fireworks-ai" - -deploymentLabels: - bu: central - env: prod - team: devops - priority: p1 - priority_v2: sp1 - primary_owner: deep.shah - secondary_owner: ramiz.mehran - service: fireworks-ai - service_type: producer-httpstateless - -podLabels: - bu: central - env: prod - team: devops - priority: p1 - priority_v2: sp1 - primary_owner: deep.shah - secondary_owner: ramiz.mehran - service: fireworks-ai - service_type: producer-httpstateless - -podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: "8080" - prometheus.io/scrape: "true" - telegraf.influxdata.com/class: infra - -podSecurityContext: - fsGroup: 65534 - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - -securityContext: - capabilities: - drop: - - ALL - readOnlyRootFilesystem: false - runAsNonRoot: true - runAsUser: 65534 - -service: - type: ClusterIP - port: 8080 - -httpProxy: - enabled: true -createContourGateway: true -namespace: fireworks-ai -contourResponseTimeout: false -ingress: - enabled: false - ingressClassName: contour-internal-1 - servicePortNumber: 8080 - enableWebsocket: false - hosts: - - host: fireworks-ai.prd.meesho.int - paths: - - path: / - pathType: ImplementationSpecific - slowStart: - enabled: false - aggression: 1 - minPercent: 10 - window: 120s - -resources: - limits: - cpu: "4" - memory: 8Gi - requests: - cpu: "1" - memory: 2Gi - -livenessProbe: - httpGet: - path: /health - port: http - scheme: HTTP - initialDelaySeconds: 15 - periodSeconds: 5 - timeoutSeconds: 3 - failureThreshold: 5 - successThreshold: 1 - -readinessProbe: - httpGet: - path: /health - port: http - scheme: HTTP - initialDelaySeconds: 15 - periodSeconds: 5 - timeoutSeconds: 3 - failureThreshold: 5 - successThreshold: 1 - -autoscaling: - enabled: false - -nodeSelector: - dedicated: megatetralite - -tolerations: - - key: dedicated - operator: Equal - value: megatetralite - effect: NoSchedule - -affinity: {} - -strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 100% - maxUnavailable: 0 - -terminationGracePeriodSeconds: 300 -lifecycle: - preStop: - exec: - command: - - /bin/sh - - -c - - sleep 120 - -bifrost: - appDir: /app/data - port: 8080 - host: 0.0.0.0 - logLevel: warn - logStyle: json - - client: - dropExcessRequests: false - initialPoolSize: 300 - allowedOrigins: - - "*" - enableLogging: true - disableContentLogging: false - disableDbPingsInHealth: false - dumpErrorsInConsoleLogs: false - logRetentionDays: 365 - enforceGovernanceHeader: false - enforceAuthOnInference: true - allowDirectKeys: false - maxRequestBodySizeMb: 100 - -storage: - mode: postgres - configStore: - enabled: true - logsStore: - enabled: true - -# PostgreSQL is managed by this Helm release as a separate Deployment. -postgresql: - enabled: true - external: - enabled: false - image: - repository: docker.io/library/postgres - tag: "16-alpine" - pullPolicy: IfNotPresent - auth: - username: bifrost - database: bifrost - existingSecret: fireworks-ai-vault - passwordKey: BIFROST_POSTGRES_PASSWORD - primary: - persistence: - enabled: true - size: 100Gi - resources: - limits: - cpu: "1" - memory: 2Gi - requests: - cpu: 250m - memory: 512Mi - podSecurityContext: - fsGroup: 999 - containerSecurityContext: {} - nodeSelector: - dedicated: megatetralite - tolerations: - - key: dedicated - operator: Equal - value: megatetralite - effect: NoSchedule - affinity: {} - -vectorStore: - enabled: false - type: none - -env: - - name: TZ - value: "Asia/Kolkata" - - name: TELEGRAF_UDP_HOST - valueFrom: - fieldRef: - fieldPath: status.podIP - - name: NODE_IP - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - name: POD_NAME - valueFrom: - fieldRef: - fieldPath: metadata.name - - name: POD_NAMESPACE - valueFrom: - fieldRef: - fieldPath: metadata.namespace - - name: POD_IP - valueFrom: - fieldRef: - fieldPath: status.podIP - -podDisruptionBudget: - enabled: true - maxUnavailable: "10%" - -# Reuses the current production Vault path until a dedicated path is provisioned. -externalSecret: - enabled: true - secretName: fireworks-ai-vault - path: "prd/cntr/devop/ai-gateway" - refreshInterval: "0" - secretStoreRef: vault-backend - -keda: - enabled: true - pollingInterval: 30 - minReplicaCount: 2 - maxReplicaCount: 200 - scaledown: - stabilizationWindowSeconds: 1800 - selectpolicy: Min - policies: - - type: Pods - value: 2 - periodseconds: 15 - scaleup: - stabilizationWindowSeconds: 120 - selectpolicy: Max - policies: - - type: Pods - value: 2 - periodseconds: 15 - - type: Percent - value: 10 - periodseconds: 15 - triggers: - - type: cpu - metricType: Utilization - metadata: - value: "40" diff --git a/helm-overrides/k8s-central-prd-ase1/flagger/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/flagger/custom-values.yaml deleted file mode 100644 index e3bd8a0..0000000 --- a/helm-overrides/k8s-central-prd-ase1/flagger/custom-values.yaml +++ /dev/null @@ -1,68 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - dedicated: central-devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: central-devops - bu: infra - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-central-rollout-service.prd-central-rollout-service.svc.cluster.local" - - -rolloutService: - enabled: true - apps: - prd-devops-node-app-consumers: true - prd-devops-node-app: true - prd-devops-golang-app: true - prd-devops-golang-app-test: true - prd-devops-golang-app-crons: true diff --git a/helm-overrides/k8s-central-prd-ase1/fluentd-copy/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/fluentd-copy/custom-values.yaml deleted file mode 100644 index 58bdf35..0000000 --- a/helm-overrides/k8s-central-prd-ase1/fluentd-copy/custom-values.yaml +++ /dev/null @@ -1,787 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-cntr-cnsre-fluentd-prd@meesho-central-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: node_pool - operator: In - values: - - np-cntr-cndvs-blue-amd-prd-ase1 - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: central - team: sre - type: fluentd - service: fluentd-copy-central-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: central - team: sre - type: fluentd - service: fluentd-copy-central-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: debug -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: [] -# - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-central-prd-ase1/fluentd/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/fluentd/custom-values.yaml deleted file mode 100644 index 8029a33..0000000 --- a/helm-overrides/k8s-central-prd-ase1/fluentd/custom-values.yaml +++ /dev/null @@ -1,829 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-cntr-cnsre-fluentd-prd@meesho-central-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: node_pool - operator: NotIn - values: - - np-cntr-cndvs-blue-amd-prd-ase1 - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: central - team: sre - type: fluentd - service: fluentd-central-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: central - team: sre - type: fluentd - service: fluentd-central-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: - - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - from_encoding "utf-8" - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-central-prd-ase1/keda/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/keda/custom-values.yaml deleted file mode 100644 index 3824a00..0000000 --- a/helm-overrides/k8s-central-prd-ase1/keda/custom-values.yaml +++ /dev/null @@ -1,42 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: central-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - podLabels: - bu: "central" - team: "central-devops" - metricsAdapter: - bu: "central" - team: "central-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 200m - memory: 200Mi - metricServer: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 200m - memory: 200Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 50m - memory: 150Mi - requests: - cpu: 10m - memory: 80Mi diff --git a/helm-overrides/k8s-central-prd-ase1/kube-dns/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/kube-dns/custom-values.yaml deleted file mode 100644 index 377e4f7..0000000 --- a/helm-overrides/k8s-central-prd-ase1/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.137.36.2"],"prd.mrouter.int.svc.cluster.local":["10.137.36.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.137.36.2"]} diff --git a/helm-overrides/k8s-central-prd-ase1/kube-events/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/kube-events/custom-values.yaml deleted file mode 100644 index 0dea214..0000000 --- a/helm-overrides/k8s-central-prd-ase1/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-central-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - dedicated: central-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: central-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: central-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: central-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: central-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: central-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-central-prd-ase1/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 823685e..0000000 --- a/helm-overrides/k8s-central-prd-ase1/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-central-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-central-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "central" - team: "central-sre" - service: "kube-state-metrics-central-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "vmselect-mds" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 23m - memory: 169Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-central-prd-ase1/kubectl-mcp-server/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/kubectl-mcp-server/custom-values.yaml deleted file mode 100644 index 20e88dd..0000000 --- a/helm-overrides/k8s-central-prd-ase1/kubectl-mcp-server/custom-values.yaml +++ /dev/null @@ -1,144 +0,0 @@ -fullnameOverride: "kubectl-mcp-server" - -replicas: 1 - -image: - # In-house hardened build (helm-templates/kubectl-mcp-server/docker/) — - # NOT the upstream Docker Hub image. 266→18 HIGH/CRIT vulns, non-root, - # multi-stage slim base, kubectl v1.33.12 / helm v3.21.0. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/prd/devop/kubectl-mcp-server - tag: "v2" - pullPolicy: IfNotPresent - -labels: - bu: infra - team: devops - service: kubectl-mcp-server - env: prd - -serviceAccount: - create: true - annotations: {} - -rbac: - create: true - -podAnnotations: {} - -podSecurityContext: {} - -securityContext: - runAsNonRoot: true - runAsUser: 1000 - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -priorityClassName: "" - -nodeSelector: - dedicated: "central-devops" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "central-devops" - effect: NoSchedule - -affinity: {} - -resources: - requests: - cpu: 250m - memory: 512Mi - limits: - cpu: 1000m - memory: 1Gi - -mcp: - mode: multi - transport: http - host: "0.0.0.0" - port: 8000 - -auth: - allowAnonymous: false - -externalSecrets: - enabled: true - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/prd/cntr/devop/kubectl-mcp-server" - -federation: - forwardTimeout: 30 - clusters: - - name: self - self: true - - name: admin - endpoint: http://kubectl-mcp-server-admin.prd.meesho.int/mcp - tokenEnv: ADMIN_MCP_TOKEN - - name: dataengg - endpoint: http://kubectl-mcp-server-dataengg.prd.meesho.int/mcp - tokenEnv: DATAENGG_MCP_TOKEN - - name: datascience - endpoint: http://kubectl-mcp-server-datascience.prd.meesho.int/mcp - tokenEnv: DATASCIENCE_MCP_TOKEN - - name: demand - endpoint: http://kubectl-mcp-server-demand.prd.meesho.int/mcp - tokenEnv: DEMAND_MCP_TOKEN - - name: dsgpu - endpoint: http://kubectl-mcp-server-dsgpu.prd.meesho.int/mcp - tokenEnv: DSGPU_MCP_TOKEN - - name: farmiso - endpoint: http://kubectl-mcp-server-farmiso.prd.meesho.int/mcp - tokenEnv: FARMISO_MCP_TOKEN - - name: supply - endpoint: http://kubectl-mcp-server-supply.prd.meesho.int/mcp - tokenEnv: SUPPLY_MCP_TOKEN - -# tcpSocket on purpose — streamable-http transport exposes only /mcp, -# there is no /health route (see chart values.yaml note). -livenessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - -readinessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - -service: - port: 8000 - type: ClusterIP - -# Contour standard (2.0.0-style): renders HTTPProxy parent/child + -intra -# instead of a networking.k8s.io Ingress. -createContourGateway: true - -ingress: - enabled: true - ingressClassName: contour-internal-1 - servicePortNumber: 8000 - hosts: - - host: kubectl-mcp-server-central.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 diff --git a/helm-overrides/k8s-central-prd-ase1/kyverno/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/kyverno/custom-values.yaml deleted file mode 100644 index 947c9be..0000000 --- a/helm-overrides/k8s-central-prd-ase1/kyverno/custom-values.yaml +++ /dev/null @@ -1,2240 +0,0 @@ -global: - - # -- Internal settings used with `helm template` to generate install manifest - # @ignored - templating: - enabled: false - debug: false - version: ~ - - image: - # -- (string) Global value that allows to set a single image registry across all deployments. - # When set, it will override any values set under `.image.registry` across the chart. - registry: ~ - # -- (list) Global list of Image pull secrets - # When set, it will override any values set under `imagePullSecrets` under different components across the chart. - imagePullSecrets: [] - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - caCertificates: - # -- Global CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - # Individual controller values will override this global value - data: ~ - - # -- Global value to set single volume to be mounted for CA certificates for all deployments. - # Not used when `.Values.global.caCertificates.data` is defined - # Individual controller values will override this global value - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Additional container environment variables to apply to all containers and init containers - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Global node labels for pod assignment. Non-global values will override the global value. - nodeSelector: - dedicated: central-kyverno - - # -- Global List of node taints to tolerate. Non-global values will override the global value. - tolerations: - - key: "dedicated" - operator: "Equal" - value: "central-kyverno" - effect: "NoSchedule" - -# -- (string) Override the name of the chart -nameOverride: ~ - -# -- (string) Override the expanded name of the chart -fullnameOverride: ~ - -# -- (string) Override the namespace the chart deploys to -namespaceOverride: ~ - -upgrade: - # -- Upgrading from v2 to v3 is not allowed by default, set this to true once changes have been reviewed. - fromV2: false - -apiVersionOverride: - # -- (string) Override api version used to create `PodDisruptionBudget`` resources. - # When not specified the chart will check if `policy/v1/PodDisruptionBudget` is available to - # determine the api version automatically. - podDisruptionBudget: ~ - -rbac: - roles: - # -- Aggregate ClusterRoles to Kubernetes default user-facing roles. For more information, see [User-facing roles](https://kubernetes.io/docs/reference/access-authn-authz/rbac/#user-facing-roles) - aggregate: - admin: true - view: true - -# Use openreports.io as the API group for reporting -openreports: - # -- Enable OpenReports feature in controllers - enabled: false - # -- Whether to install CRDs from the upstream OpenReports chart. Setting this to true requires enabled to also be true. - installCrds: false - -# CRDs configuration -crds: - - # -- Whether to have Helm install the Kyverno CRDs, if the CRDs are not installed by Helm, they must be added before policies can be created - install: true - - reportsServer: - # -- Kyverno reports-server is used in your cluster - enabled: false - - groups: - - # -- Install CRDs in group `kyverno.io` - kyverno: - cleanuppolicies: true - clustercleanuppolicies: true - clusterpolicies: true - globalcontextentries: true - policies: true - policyexceptions: true - updaterequests: true - - # -- Install CRDs in group `policies.kyverno.io` - policies: - validatingpolicies: true - policyexceptions: true - imagevalidatingpolicies: true - namespacedimagevalidatingpolicies: true - mutatingpolicies: true - generatingpolicies: true - deletingpolicies: true - namespaceddeletingpolicies: true - namespacedvalidatingpolicies: true - - # -- Install CRDs in group `reports.kyverno.io` - reports: - clusterephemeralreports: true - ephemeralreports: true - - # -- Install CRDs in group `wgpolicyk8s.io` - wgpolicyk8s: - clusterpolicyreports: false ##lk - policyreports: false ##lk - - # -- Additional CRDs annotations (Replace=true avoids 262144-byte annotation limit on large CRDs) - annotations: - argocd.argoproj.io/sync-options: Replace=true - # strategy.spinnaker.io/replace: 'true' - - # -- Additional CRDs labels - customLabels: {} - - migration: - - # -- Enable CRDs migration using helm post upgrade hook - enabled: true - - # -- Resources to migrate - resources: - - cleanuppolicies.kyverno.io - - clustercleanuppolicies.kyverno.io - - clusterpolicies.kyverno.io - - globalcontextentries.kyverno.io - - policies.kyverno.io - - policyexceptions.kyverno.io - - updaterequests.kyverno.io - - deletingpolicies.policies.kyverno.io - - generatingpolicies.policies.kyverno.io - - imagevalidatingpolicies.policies.kyverno.io - - namespacedimagevalidatingpolicies.policies.kyverno.io - - mutatingpolicies.policies.kyverno.io - - namespaceddeletingpolicies.policies.kyverno.io - - namespacedvalidatingpolicies.policies.kyverno.io - - policyexceptions.policies.kyverno.io - - validatingpolicies.policies.kyverno.io - - image: - # -- (string) Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- (string) Image repository - repository: kyverno/kyverno-cli - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- (string) Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podResources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -# Configuration -config: - - # -- Create the configmap. - create: true - - # -- Preserve the configmap settings during upgrade. - preserve: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - # -- Enable registry mutation for container images. Enabled by default. - enableDefaultRegistryMutation: true - - # -- The registry hostname used for the image mutation. - defaultRegistry: docker.io - - # -- Exclude groups - excludeGroups: - - system:nodes - - # -- Exclude usernames - excludeUsernames: [] - # - '!system:kube-scheduler' - - # -- Exclude roles - excludeRoles: [] - - # -- Exclude roles - excludeClusterRoles: [] - - # -- Generate success events. - generateSuccessEvents: false - - # -- Resource types to be skipped by the Kyverno policy engine. - # Make sure to surround each entry in quotes so that it doesn't get parsed as a nested YAML list. - # These are joined together without spaces, run through `tpl`, and the result is set in the config map. - # @default -- See [values.yaml](values.yaml) - resourceFilters: - - '[Event,*,*]' - # - '[*/*,kube-system,*]' - - '[*/*,kube-public,*]' - - '[*/*,kube-node-lease,*]' - - '[Node,*,*]' - - '[Node/?*,*,*]' - - '[APIService,*,*]' - - '[APIService/?*,*,*]' - - '[TokenReview,*,*]' - - '[SubjectAccessReview,*,*]' - - '[SelfSubjectAccessReview,*,*]' - - '[Binding,*,*]' - - '[Pod/binding,*,*]' - - '[ReplicaSet,*,*]' - - '[ReplicaSet/?*,*,*]' - - '[EphemeralReport,*,*]' - - '[ClusterEphemeralReport,*,*]' - # exclude resources from the chart - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:additional]' - - '[ClusterRoleBinding,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.configMapName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.metricsConfigMapName" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Job,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[Job/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.admission-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.background-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - # -- Sets the threshold for the total number of UpdateRequests generated for mutateExisitng and generate policies. - updateRequestThreshold: 1000 - - # -- Defines the `namespaceSelector`/`objectSelector` in the webhook configurations. - # The Kyverno namespace is excluded if `excludeKyvernoNamespace` is `true` (default) - webhooks: - # Exclude namespaces - # namespaceSelector: lk-s2 - # matchExpressions: - # - key: kubernetes.io/metadata.name - # operator: NotIn - # values: - # - kube-systemm lk-s2 - # Exclude objects - # objectSelector: - # matchExpressions: - # - key: webhooks.kyverno.io/exclude - # operator: DoesNotExist - - # -- Defines annotations to set on webhook configurations. - webhookAnnotations: - # Example to disable admission enforcer on AKS: - # 'admissions.enforcer/disabled': 'true' lk-s3 - - # -- Defines labels to set on webhook configurations. - webhookLabels: {} - # Example to adopt webhook resources in ArgoCD: - # 'argocd.argoproj.io/instance': 'kyverno' - - # -- Defines match conditions to set on webhook configurations (requires Kubernetes 1.27+). - matchConditions: [] - - # -- Exclude Kyverno namespace - # Determines if default Kyverno namespace exclusion is enabled for webhooks and resourceFilters - excludeKyvernoNamespace: true - - # -- resourceFilter namespace exclude - # Namespaces to exclude from the default resourceFilters - resourceFiltersExcludeNamespaces: [] - - # -- resourceFilters exclude list - # Items to exclude from config.resourceFilters - resourceFiltersExclude: [] - - # -- resourceFilter namespace include - # Namespaces to include to the default resourceFilters - resourceFiltersIncludeNamespaces: [] - - # -- resourceFilters include list - # Items to include to config.resourceFilters - resourceFiltersInclude: [] - -# Metrics configuration -metricsConfig: - - # -- Create the configmap. - create: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - namespaces: - - # -- List of namespaces to capture metrics for. - include: [] - - # -- list of namespaces to NOT capture metrics for. - exclude: [] - - # -- (string) Rate at which metrics should reset so as to clean up the memory footprint of kyverno metrics, if you might be expecting high memory footprint of Kyverno's metrics. Default: 0, no refresh of metrics. WARNING: This flag is not working since Kyverno 1.8.0 - metricsRefreshInterval: ~ - # metricsRefreshInterval: 24h - - # -- (list) Configures the bucket boundaries for all Histogram metrics, changing this configuration requires restart of the kyverno admission controller - bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 15, 20, 25, 30] - - # -- (map) Configures the exposure of individual metrics, by default all metrics and all labels are exported, changing this configuration requires restart of the kyverno admission controller - metricsExposure: - kyverno_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_image_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_mutating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_generating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_admission_review_duration_seconds: - # enabled: false - disabledLabelDimensions: ["resource_namespace"] - kyverno_policy_rule_info_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_policy_results_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_admission_requests_total: - disabledLabelDimensions: ["resource_namespace"] - kyverno_cleanup_controller_deletedobjects_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - -# -- Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -imagePullSecrets: {} - # regcred: - # registry: foo.example.com - # username: foobar - # password: secret - # regcred2: - # registry: bar.example.com - # username: barbaz - # password: secret2 - -# -- Existing Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -existingImagePullSecrets: [] - # - test-registry - # - other-test-registry - -# Tests configuration -test: - # -- Sleep time before running test - sleep: 20 - - image: - # -- (string) Image registry - registry: curlimages - # -- Image repository - repository: curl - # -- Image tag - # Defaults to `latest` if omitted - tag: '8.10.1' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - # -- Security context for the test containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- Additional Pod annotations - podAnnotations: {} - - # -- List of node taints to tolerate - tolerations: [] - -# -- Additional labels -customLabels: {} - - -webhooksCleanup: - # -- Create a helm pre-delete hook to cleanup webhooks. - enabled: true - - autoDeleteWebhooks: - # -- Allow webhooks controller to delete webhooks using finalizers - enabled: false - - image: - # -- (string) Image registry - registry: registry.k8s.io - # -- Image repository - repository: kubectl - # -- Image tag - # Defaults to `latest` if omitted - tag: 'v1.32.7' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -grafana: - # -- Enable grafana dashboard creation. - enabled: false - - # -- Configmap name template. - configMapName: '{{ include "kyverno.fullname" . }}-grafana' - - # -- (string) Namespace to create the grafana dashboard configmap. - # If not set, it will be created in the same namespace where the chart is deployed. - namespace: ~ - - # -- Grafana dashboard configmap annotations. - annotations: {} - - # -- Grafana dashboard configmap labels - labels: - grafana_dashboard: "1" - - # -- create GrafanaDashboard custom resource referencing to the configMap. - # according to https://grafana-operator.github.io/grafana-operator/docs/examples/dashboard_from_configmap/readme/ - grafanaDashboard: - create: false - folder: kyverno - allowCrossNamespaceImport: true - matchLabels: - dashboards: "grafana" - -# Features configuration -features: - admissionReports: - # -- Enables the feature - enabled: true - aggregateReports: - # -- Enables the feature - enabled: true - policyReports: - # -- Enables the feature - enabled: false ##lk - validatingAdmissionPolicyReports: - # -- Enables the feature - enabled: true - mutatingAdmissionPolicyReports: - # -- Enables the feature - enabled: false - reporting: - # -- Enables the feature - validate: true - # -- Enables the feature - mutate: true - # -- Enables the feature - mutateExisting: true - # -- Enables the feature - imageVerify: true - # -- Enables the feature - generate: true - autoUpdateWebhooks: - # -- Enables the feature - enabled: true - backgroundScan: - # -- Enables the feature - enabled: true - # -- Number of background scan workers - backgroundScanWorkers: 2 - # -- Background scan interval - backgroundScanInterval: 1h - # -- Skips resource filters in background scan - skipResourceFilters: true - configMapCaching: - # -- Enables the feature - enabled: true - controllerRuntimeMetrics: - # -- Bind address for controller-runtime metrics (use "0" to disable it) - bindAddress: ":8080" - deferredLoading: - # -- Enables the feature - enabled: true - dumpPayload: - # -- Enables the feature - enabled: false - forceFailurePolicyIgnore: - # -- Enables the feature - enabled: false - generateValidatingAdmissionPolicy: - # -- Enables the feature - enabled: true - generateMutatingAdmissionPolicy: - # -- Enables the feature - enabled: false - dumpPatches: - # -- Enables the feature - enabled: false - globalContext: - # -- Maximum allowed response size from API Calls. A value of 0 bypasses checks (not recommended) - maxApiCallResponseLength: 2000000 - logging: - # -- Logging format - format: text - # -- Logging verbosity - verbosity: 2 - omitEvents: - # -- Events which should not be emitted (possible values `PolicyViolation`, `PolicyApplied`, `PolicyError`, and `PolicySkipped`) - eventTypes: - - PolicyApplied - - PolicySkipped - # - PolicyViolation - # - PolicyError - policyExceptions: - # -- Enables the feature - enabled: true - # -- Restrict policy exceptions to a single namespace - # Set to "*" to allow exceptions in all namespaces - namespace: '' - protectManagedResources: - # -- Enables the feature - enabled: false - registryClient: - # -- Allow insecure registry - allowInsecure: false - # -- Enable registry client helpers - credentialHelpers: - - default - - google - - amazon - - azure - - github - ttlController: - # -- Reconciliation interval for the label based cleanup manager - reconciliationInterval: 1m - tuf: - # -- Enables the feature - enabled: false - # -- (string) Path to Tuf root - root: ~ - # -- (string) Raw Tuf root - rootRaw: ~ - # -- (string) Tuf mirror - mirror: ~ - -# Admission controller configuration -admissionController: - autoscaling: - # -- Enable horizontal pod autoscaling - enabled: true - - # -- Minimum number of pods - minReplicas: 1 - - # -- Maximum number of pods - maxReplicas: 10 - - # -- Target CPU utilization percentage - targetCPUUtilizationPercentage: 80 - - # -- Configurable scaling behavior - behavior: {} - - # -- Overrides features defined at the root level - featuresOverride: - admissionReports: - # -- Max number of admission reports allowed in flight until the admission controller stops creating new ones - backPressureThreshold: 1000 - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- The ServiceAccount name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno admission controller activities. - # This will help ensure Kyverno stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- admissionController webhook server port - # in case you are using hostNetwork: true, you might want to change the port the webhookServer is listening to - webhookServer: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - admission-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.admissionController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - initContainer: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyvernopre - # -- (string) Image tag - # If missing, defaults to image.tag - tag: ~ - # -- (string) Image pull policy - # If missing, defaults to image.pullPolicy - pullPolicy: ~ - - resources: - # -- Pod resource limits - # limits: - # cpu: 100m - # memory: 256Mi - # -- Pod resource requests - requests: - cpu: 1000m - memory: 1Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - container: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyverno - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - resources: - # -- Pod resource limits - # limits: - # memory: 384Mi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Array of extra init containers - extraInitContainers: [] - # - name: init-container - # image: busybox - # command: ['sh', '-c', 'echo Hello'] - - # -- Array of extra containers to run alongside kyverno - extraContainers: [] - # - name: myapp-container - # image: busybox - # command: ['sh', '-c', 'echo Hello && sleep 3600'] - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Kyverno's metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Background controller configuration -backgroundController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable background controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: - - apiGroups: - - networking.k8s.io - resources: - - ingresses - - ingressclasses - - networkpolicies - verbs: - - create - - update - - patch - - delete - - apiGroups: - - rbac.authorization.k8s.io - resources: - - rolebindings - - roles - verbs: - - create - - update - - patch - - delete - - apiGroups: - - '' - resources: - - configmaps - - resourcequotas - - limitranges - verbs: - - create - - update - - patch - - delete - - apiGroups: - - resource.k8s.io - resources: - - resourceclaims - - resourceclaimtemplates - verbs: - - create - - delete - - update - - patch - - deletecollection - - apiGroups: - - autoscaling - resources: - - horizontalpodautoscalers - verbs: - - get - - list - - watch - - update - - patch - - apiGroups: - - apps - resources: - - deployments - - statefulsets - - replicasets - verbs: - - get - - list - - watch - - update - - patch - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - # - patch - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/background-controller - # -- Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - # limits: - # memory: 128Mi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - background-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.backgroundController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - # -- backgroundController server port - # in case you are using hostNetwork: true, you might want to change the port the backgroundController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Cleanup controller configuration -cleanupController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable cleanup controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - delete - # - list - # - watch - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/cleanup-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- cleanupController server port - # in case you are using hostNetwork: true, you might want to change the port the cleanupController is listening to - server: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - # limits: - # memory: 128Mi - # -- Pod resource requests - requests: - cpu: 1100m - memory: 1.2Gi - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - cleanup-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `service.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Reports controller configuration -reportsController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable reports controller. - enabled: false ##lk - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/reports-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno reports controller activities. - # This will help ensure Kyverno reports stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - memory: 128Mi - # -- Pod resource requests - requests: - cpu: 100m - memory: 64Mi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - reports-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.reportsController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- (string) Service node port. - # Only used if `type` is `NodePort`. - nodePort: ~ - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- (string) Traces receiver address - address: ~ - # -- (string) Traces receiver port - port: ~ - # -- (string) Traces receiver credentials - creds: ~ - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- (string) Otel collector endpoint - collector: ~ - # -- (string) Otel collector credentials - creds: ~ - - # -- reportsController server port - # in case you are using hostNetwork: true, you might want to change the port the reportsController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - - # -- Enable sanity check for reports CRDs - sanityChecks: true diff --git a/helm-overrides/k8s-central-prd-ase1/kyverno/policies/protect-namespace.yaml b/helm-overrides/k8s-central-prd-ase1/kyverno/policies/protect-namespace.yaml deleted file mode 100644 index 54f5162..0000000 --- a/helm-overrides/k8s-central-prd-ase1/kyverno/policies/protect-namespace.yaml +++ /dev/null @@ -1,40 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: protect-critical-namespaces -spec: - validationFailureAction: Enforce - background: false - rules: - - name: deny-deletion-of-critical-namespaces - match: - any: - - resources: - kinds: - - Namespace - names: - - keda-central-prd - - contour-internal-0-central-prd - - contour-internal-0-central-prd-intra - - contour-external-central-prd - - external-secrets-central-prd - - flagger-central-prd - - victoriametrics - - kube-system - - monitoring - - telegraf-operator - - loadtester - - opentelemetry - - fluentd - - kube-events - - observability - - contour-cert-checker-ns - - aurva-dataplane - validate: - message: "Deletion of critical namespaces is not allowed." - deny: - conditions: - any: - - key: "{{request.operation}}" - operator: Equals - value: DELETE diff --git a/helm-overrides/k8s-central-prd-ase1/kyverno/policies/restrict-replicas.yaml b/helm-overrides/k8s-central-prd-ase1/kyverno/policies/restrict-replicas.yaml deleted file mode 100644 index d456985..0000000 --- a/helm-overrides/k8s-central-prd-ase1/kyverno/policies/restrict-replicas.yaml +++ /dev/null @@ -1,29 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: restrict-minimum-replicas-selected-ns -spec: - validationFailureAction: Enforce - background: false - rules: - - name: min-replicas-deployments-statefulsets - match: - resources: - kinds: - - Deployment - - StatefulSet - - Deployment/scale - - StatefulSet/scale - names: - - coredns - namespaces: - - kube-system - operations: - - CREATE - - UPDATE - validate: - message: "Deployments and StatefulSets must have at least 5 replicas in this namespace." - anyPattern: - # Case 1: replicas explicitly set and >= 5 - - spec: - replicas: ">=5" diff --git a/helm-overrides/k8s-central-prd-ase1/loadtester/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/loadtester/custom-values.yaml deleted file mode 100644 index f5fdbfe..0000000 --- a/helm-overrides/k8s-central-prd-ase1/loadtester/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -replicaCount: 1 - -Namespace: loadtester - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/dev/devops/flagger-loadtester - tag: 0.30.0 - pullPolicy: IfNotPresent - pullSecret: - -podLabels: - bu: central - team: central-shared - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - -podPriorityClassName: "" - -logLevel: info -cmd: - timeout: 1h - namespaceRegexp: "" - -nameOverride: "loadtester" -fullnameOverride: "" - -env: [] - -service: - type: ClusterIP - port: 80 - -resources: - requests: - cpu: 10m - memory: 64Mi - -volumes: [] -volumeMounts: [] - -nodeSelector: - dedicated: central-devops - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - -affinity: {} - -ingress: - slowStart: - enabled: false - enabled: true - ingressClassName: contour-internal-1 - servicePort: http - hosts: - - host: central-flagger-loadtester.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / -createContourGateway: true -contourResponseTimeout: false -rbac: - # rbac.create: `true` if rbac resources should be created - create: true - # rbac.scope: `cluster` to create cluster-scope rbac resources (ClusterRole/ClusterRoleBinding) - # otherwise, namespace-scope rbac resources will be created (Role/RoleBinding) - scope: - # rbac.rules: array of rules to apply to the role. example: - # rules: - # - apiGroups: [""] - # resources: ["pods"] - # verbs: ["list", "get"] - rules: [] - -# name of an existing service account to use - if not creating rbac resources -serviceAccountName: "" - -# App Mesh virtual node settings (to be used for AppMesh v1beta1) -meshName: "" -#backends: -# - app1.namespace -# - app2.namespace - -# App Mesh virtual node settings (to be used for AppMesh v1beta2) -appmesh: - enabled: false - backends: - - podinfo - - podinfo-canary - -#Istio virtual service and gatway settings. TLS secrets should be in namespace before enbaled it. ( secret format loadtester.fullname ) -istio: - enabled: false - host: flagger-loadtester.flagger - gateway: - enabled: false - tls: - enabled: false - httpsRedirect: false - -# when enabled, it will add a security context for the loadtester pod -securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 100 - runAsGroup: 101 - -podDisruptionBudget: - enabled: false - minAvailable: 1 diff --git a/helm-overrides/k8s-central-prd-ase1/opentelemetry-claude-metrics/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/opentelemetry-claude-metrics/custom-values.yaml deleted file mode 100644 index 5b6fdde..0000000 --- a/helm-overrides/k8s-central-prd-ase1/opentelemetry-claude-metrics/custom-values.yaml +++ /dev/null @@ -1,338 +0,0 @@ -nameOverride: "" -fullnameOverride: "opentelemetry-claude-metrics" - -additionalLabels: - bu: "central" - team: "sre" - service: "opentelemetry-claude-metrics" - env: "prd" - priority: "p1" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: true - key: meesho/prd/cntr/xcntr/otel-claude-metrics - secretStoreRef: - name: vault-backend - -mode: "deployment" - -namespaceOverride: "" - -presets: - logsCollection: - enabled: false - hostMetrics: - enabled: false - kubernetesAttributes: - enabled: false - kubeletMetrics: - enabled: false - kubernetesEvents: - enabled: false - clusterMetrics: - enabled: false - -configMap: - create: true - -config: - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - auth: - authenticator: bearertokenauth - http: - endpoint: ${env:MY_POD_IP}:4318 - auth: - authenticator: bearertokenauth - - processors: - batch: - send_batch_size: 1024 - send_batch_max_size: 2048 - timeout: 10s - memory_limiter: - check_interval: 1s - limit_percentage: 85 - spike_limit_percentage: 20 - - # --- PII redaction for Cowork event bodies (prompt text) -------------- - # Cowork events carry raw prompt text. This drops it BEFORE the events - # reach ClickHouse, while keeping cost/token/tool fields for stats. - # error_mode: ignore => a wrong key is a silent no-op (won't crash the - # pipeline), which ALSO means: do NOT assume prompts are redacted until - # you confirm the real key/location from the debug output. - # - If prompt text is a log ATTRIBUTE -> delete_key(attributes, "") - # - If it is in the log BODY -> set(body, "") (uncomment below) - # Confirm the exact key ("prompt", "prompt.text", …) from step-1 debug logs. - transform/redact: - error_mode: ignore - log_statements: - - context: log - statements: - # - delete_key(attributes, "prompt") # commented: allow prompt text through for observability - # - delete_key(attributes, "prompt.text") # commented: allow prompt text through for observability - # - set(body, "") where attributes["event.name"] == "user_prompt" - - # Pin the HELP (description) string for claude_code.lines_of_code.count so - # a mixed fleet (CLI < 2.1.172 emits the old description; >= 2.1.172 emits - # the new one) doesn't trigger the prometheus exporter's "N error(s) - # occurred for claude_observability_claude_code_lines_of_code_count_total" - # HELP-collision error. If the same pattern shows up on other metrics - # (token.usage, cost.usage, ...), add another `where name == "..."` line - # here rather than defining a second processor. - transform/claude_metrics_help_fix: - error_mode: ignore - metric_statements: - - context: metric - statements: - - set(description, "Count of lines of code modified, with the 'type' attribute indicating whether lines were added or removed and the 'model' attribute indicating which model made the change") where name == "claude_code.lines_of_code.count" - - exporters: - # Claude Code metrics path — unchanged. - prometheus: - endpoint: "0.0.0.0:8889" - namespace: claude_observability - resource_to_telemetry_conversion: - enabled: true - metric_expiration: 5m - - # Cowork events land here as raw OTel logs. - # NOTE: verify DSN scheme (tcp:// vs clickhouse://) and the `ttl` field - # name against the clickhouseexporter README for image tag 0.111.0 — - # both changed across releases and are the most likely mismatch. - clickhouse: - endpoint: http://claude-observability.prd.meesho.int:8080?dial_timeout=10s&compress=lz4 # ClickHouse HTTP interface (DNS — VM IP is not stable) - database: otel - username: claude - password: ${CLICKHOUSE_PASSWORD} # must be injected via the vault secret (see note) - logs_table_name: otel_logs - ttl: 720h # 30d retention, enforced by ClickHouse - create_schema: true # needs CREATE priv on the writer role; else set false + pre-create table - async_insert: true - timeout: 10s - sending_queue: - queue_size: 5000 - retry_on_failure: - enabled: true - - # TEMPORARY: prints full event bodies (incl. prompt text) to stdout. - # Keep for first-deploy validation only, then remove from the logs pipeline. - debug: - verbosity: detailed - - extensions: - health_check: - path: /health - bearertokenauth: - token: ${AUTH_TOKEN} - - service: - telemetry: - metrics: - level: normal - address: ${env:MY_POD_IP}:8888 - extensions: - - health_check - - bearertokenauth - pipelines: - traces: null - # Cowork events arrive as OTLP logs, get redacted, and are stored raw in - # ClickHouse. Aggregation happens at query time in SQL — no connectors. - # `debug` is validation-only; remove it (and the redact caveat aside) - # once you've confirmed events land and the redaction key is correct. - logs: - receivers: - - otlp - processors: - - memory_limiter - - transform/redact - - batch - exporters: - - clickhouse - - debug - # Existing Claude Code metrics only — Cowork no longer feeds this pipeline. - metrics: - receivers: - - otlp - processors: - - memory_limiter - - transform/claude_metrics_help_fix - - batch - exporters: - - prometheus - -image: - repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - tag: "0.111.0" - digest: "" - -imagePullSecrets: [] - -command: - name: otelcol-contrib - extraArgs: [] - -serviceAccount: - create: true - annotations: {} - name: "" - -clusterRole: - create: false - annotations: {} - name: "" - rules: [] - clusterRoleBinding: - annotations: {} - name: "" - -podSecurityContext: {} -securityContext: {} - -nodeSelector: {} - -tolerations: [] - -affinity: {} -topologySpreadConstraints: [] -priorityClassName: "" - -extraEnvs: - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - - name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - -extraEnvsFrom: - - secretRef: - name: opentelemetry-claude-metrics-secret -extraVolumes: [] -extraVolumeMounts: [] - -ports: - otlp: - enabled: true - containerPort: 4317 - servicePort: 4317 - protocol: TCP - appProtocol: grpc - otlp-http: - enabled: true - containerPort: 4318 - servicePort: 4318 - protocol: TCP - prom-exporter: - enabled: true - containerPort: 8889 - servicePort: 8889 - protocol: TCP - jaeger-compact: - enabled: false - jaeger-thrift: - enabled: false - jaeger-grpc: - enabled: false - zipkin: - enabled: false - metrics: - enabled: true - containerPort: 8888 - servicePort: 8888 - protocol: TCP - -resources: - requests: - cpu: 2 - memory: 2Gi - limits: - cpu: 2 - memory: 2Gi - -podAnnotations: - otel.io/path: /metrics - otel.io/port: "8888" - otel.io/scrape: "true" - -podLabels: {} - -hostNetwork: false -dnsPolicy: "ClusterFirstWithHostNet" -dnsConfig: {} - -replicaCount: 2 -revisionHistoryLimit: 10 - -annotations: {} -extraContainers: [] -initContainers: [] -lifecycleHooks: {} - -livenessProbe: - httpGet: - port: 13133 - path: /health - -readinessProbe: - httpGet: - port: 13133 - path: /health - -service: - type: ClusterIP - annotations: - io.cilium/global-service: "true" - cloud.google.com/neg: '{"exposed_ports": {"4318":{"name": "otel-cld-ext-cntr-prd"}}}' - -ingress: - enabled: false - -podMonitor: - enabled: false - -serviceMonitor: - enabled: false - -podDisruptionBudget: - enabled: true - maxUnavailable: 1 - -autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 10 - behavior: {} - targetCPUUtilizationPercentage: 70 - targetMemoryUtilizationPercentage: 60 - -rollout: - rollingUpdate: - maxUnavailable: 1 - strategy: RollingUpdate - -prometheusRule: - enabled: false - groups: [] - defaultRules: - enabled: false - extraLabels: {} - -statefulset: - volumeClaimTemplates: [] - podManagementPolicy: "Parallel" - -networkPolicy: - enabled: false - annotations: {} - allowIngressFrom: [] - extraIngressRules: [] - egressRules: [] diff --git a/helm-overrides/k8s-central-prd-ase1/opentelemetry-codex-metrics/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/opentelemetry-codex-metrics/custom-values.yaml deleted file mode 100644 index b364842..0000000 --- a/helm-overrides/k8s-central-prd-ase1/opentelemetry-codex-metrics/custom-values.yaml +++ /dev/null @@ -1,328 +0,0 @@ -nameOverride: "" -fullnameOverride: "opentelemetry-codex-metrics" - -additionalLabels: - bu: "central" - team: "sre" - service: "opentelemetry-codex-metrics" - env: "prd" - priority: "p1" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: true - key: meesho/prd/cntr/xcntr/otel-codex-metrics - secretStoreRef: - name: vault-backend - -mode: "deployment" - -namespaceOverride: "" - -presets: - logsCollection: - enabled: false - hostMetrics: - enabled: false - kubernetesAttributes: - enabled: false - kubeletMetrics: - enabled: false - kubernetesEvents: - enabled: false - clusterMetrics: - enabled: false - -configMap: - create: true - -config: - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - auth: - authenticator: bearertokenauth - http: - endpoint: ${env:MY_POD_IP}:4318 - auth: - authenticator: bearertokenauth - - processors: - batch: - send_batch_size: 1024 - send_batch_max_size: 2048 - timeout: 10s - memory_limiter: - check_interval: 1s - limit_percentage: 85 - spike_limit_percentage: 20 - # codex.tool_result events carry raw shell command arguments and raw - # command output, which can contain secrets. Drop them before export. - # User prompts are already redacted client-side (log_user_prompt=false). - attributes/scrub-sensitive: - actions: - - key: arguments - action: delete - - key: output - action: delete - # Codex uses reasoning_effort on conversation events and - # model_reasoning_effort on completed response events. Normalize both so - # the derived metric exposes one bounded label. - transform/codex-log-attributes: - error_mode: ignore - log_statements: - - context: log - statements: - - set(attributes["codex.reasoning_effort"], attributes["reasoning_effort"]) where attributes["reasoning_effort"] != nil - - set(attributes["codex.reasoning_effort"], attributes["model_reasoning_effort"]) where attributes["codex.reasoning_effort"] == nil and attributes["model_reasoning_effort"] != nil - - connectors: - # Convert selected Codex log dimensions into a Prometheus counter. - # user.email and user.account_id are included intentionally for per-user - # usage attribution (parity with Claude Code telemetry). Do NOT add - # conversation ID, prompt, arguments, or output as labels. - count/codex_logs: - logs: - codex.log.events: - description: Count of Codex OTLP log events by bounded dimensions - conditions: - - 'attributes["event.name"] != nil' - attributes: - - key: event.name - default_value: unknown - - key: event.kind - default_value: unknown - - key: user.account_id - default_value: unknown - - key: user.email - default_value: unknown - - key: codex.reasoning_effort - default_value: unknown - - key: originator - default_value: unknown - - key: app.version - default_value: unknown - - exporters: - prometheus: - endpoint: "0.0.0.0:8889" - namespace: codex_observability - resource_to_telemetry_conversion: - enabled: true - metric_expiration: 5m - # Phase 1 log sink: registers /v1/logs on the OTLP HTTP receiver and - # surfaces traffic in collector stdout for validation. A follow-up will - # switch this to a persistent store once the backend is finalised. - debug/logs: - verbosity: basic - - extensions: - health_check: - path: /health - bearertokenauth: - token: ${AUTH_TOKEN} - - service: - telemetry: - metrics: - level: normal - address: ${env:MY_POD_IP}:8888 - extensions: - - health_check - - bearertokenauth - pipelines: - traces: null - logs: - receivers: - - otlp - processors: - - memory_limiter - - attributes/scrub-sensitive - - transform/codex-log-attributes - - batch - exporters: - - debug/logs - - count/codex_logs - metrics: - receivers: - - otlp - - count/codex_logs - processors: - - memory_limiter - - batch - exporters: - - prometheus - -image: - repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - tag: "0.111.0" - digest: "" - -imagePullSecrets: [] - -command: - name: otelcol-contrib - extraArgs: [] - -serviceAccount: - create: true - annotations: {} - name: "" - -clusterRole: - create: false - annotations: {} - name: "" - rules: [] - clusterRoleBinding: - annotations: {} - name: "" - -podSecurityContext: {} -securityContext: {} - -nodeSelector: {} - -tolerations: [] - -affinity: {} -topologySpreadConstraints: [] -priorityClassName: "" - -extraEnvs: - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - - name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - -extraEnvsFrom: - - secretRef: - name: opentelemetry-codex-metrics-secret -extraVolumes: [] -extraVolumeMounts: [] - -ports: - otlp: - enabled: true - containerPort: 4317 - servicePort: 4317 - protocol: TCP - appProtocol: grpc - otlp-http: - enabled: true - containerPort: 4318 - servicePort: 4318 - protocol: TCP - prom-exporter: - enabled: true - containerPort: 8889 - servicePort: 8889 - protocol: TCP - jaeger-compact: - enabled: false - jaeger-thrift: - enabled: false - jaeger-grpc: - enabled: false - zipkin: - enabled: false - metrics: - enabled: true - containerPort: 8888 - servicePort: 8888 - protocol: TCP - -resources: - requests: - cpu: 2 - memory: 2Gi - limits: - cpu: 2 - memory: 2Gi - -podAnnotations: - otel.io/path: /metrics - otel.io/port: "8888" - otel.io/scrape: "true" - -podLabels: {} - -hostNetwork: false -dnsPolicy: "ClusterFirstWithHostNet" -dnsConfig: {} - -replicaCount: 2 -revisionHistoryLimit: 10 - -annotations: {} -extraContainers: [] -initContainers: [] -lifecycleHooks: {} - -livenessProbe: - httpGet: - port: 13133 - path: /health - -readinessProbe: - httpGet: - port: 13133 - path: /health - -service: - type: ClusterIP - annotations: - io.cilium/global-service: "true" - cloud.google.com/neg: '{"exposed_ports": {"4318":{"name": "otel-cdx-ext-cntr-prd"}}}' - -ingress: - enabled: false - -podMonitor: - enabled: false - -serviceMonitor: - enabled: false - -podDisruptionBudget: - enabled: true - maxUnavailable: 1 - -autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 10 - behavior: {} - targetCPUUtilizationPercentage: 70 - targetMemoryUtilizationPercentage: 60 - -rollout: - rollingUpdate: - maxUnavailable: 1 - strategy: RollingUpdate - -prometheusRule: - enabled: false - groups: [] - defaultRules: - enabled: false - extraLabels: {} - -statefulset: - volumeClaimTemplates: [] - podManagementPolicy: "Parallel" - -networkPolicy: - enabled: false - annotations: {} - allowIngressFrom: [] - extraIngressRules: [] - egressRules: [] diff --git a/helm-overrides/k8s-central-prd-ase1/opentelemetry-coralogix/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/opentelemetry-coralogix/custom-values.yaml deleted file mode 100644 index 35137df..0000000 --- a/helm-overrides/k8s-central-prd-ase1/opentelemetry-coralogix/custom-values.yaml +++ /dev/null @@ -1,277 +0,0 @@ -global: - domain: "coralogixsg.com" - defaultApplicationName: "default" - defaultSubsystemName: "nodes" - - # Old endpoint based configuration, - # please use domain instead. - traces: - endpoint: "" - metrics: - endpoint: "" - logs: - endpoint: "" - -# set distribution to openshift for openshift clusters -distribution: "" - -# Opentelemetry collector configuration -dedicatedValue: false -schedulerName: default-scheduler - -labels: - bu: "central" - team: "sre-shared" - service: "opentelemetry-central-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: false - key: prd/admin/coralogix-keys - secretStoreRef: - name: vault-backend - -image: - # If you want to use the core image `otel/opentelemetry-collector`, you also need to change `command.name` value to `otelcol`. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - # repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - tag: "0.87.0" - # When digest is set to a non-empty value, images will be pulled by digest (regardless of tag value). - digest: "" -mode: daemonset -rollout: - rollingUpdate: - # When 'mode: daemonset', maxSurge cannot be used when hostPort is set for any of the ports - # maxSurge: 25% - maxUnavailable: 10 - strategy: RollingUpdate -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" -hostNetwork: true -dnsPolicy: "ClusterFirstWithHostNet" -fullnameOverride: opentelemetry-central-prd - -presets: - logsCollection: - enabled: false - storeCheckpoints: false - maxRecombineLogSize: 1048576 - extraFilelogOperators: [] -# - type: recombine -# combine_field: body -# source_identifier: attributes["log.file.path"] -# is_first_entry: body matches "^(YOUR-LOGS-REGEX)" - kubernetesAttributes: - enabled: false - hostMetrics: - enabled: false - kubeletMetrics: - enabled: false - -extraEnvs: -- name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" -- name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName -config: - extensions: {} - # zpages: - # endpoint: localhost:55679 - # pprof: - # endpoint: localhost:1777 - exporters: - loadbalancing: - protocol: - otlp: - timeout: 4s - tls: - insecure: true - sending_queue: - enabled: true - queue_size: 50000 - num_consumers: 500 - resolver: - dns: - hostname: "opentelemetry-admin-prd.opentelemetry.svc.clusterset.local" - otlp: - endpoint: opentelemetry-deployment-central-prd.opentelemetry.svc.cluster.local:4317 - tls: - insecure: true - keepalive: - timeout: 2s - # coralogix: - # timeout: "30s" - # private_key: "${CORALOGIX_PRIVATE_KEY}" - # domain: "{{.Values.global.domain}}" - # traces: - # endpoint: "{{ .Values.global.traces.endpoint }}" - # metrics: - # endpoint: "{{ .Values.global.metrics.endpoint }}" - # logs: - # endpoint: "{{ .Values.global.logs.endpoint }}" - # application_name_attributes: - # - "k8s.namespace.name" - # - "service.namespace" - # subsystem_name_attributes: - # - "k8s.deployment.name" - # - "k8s.statefulset.name" - # - "k8s.daemonset.name" - # - "k8s.cronjob.name" - # - "k8s.job.name" - # - "k8s.container.name" - # - "k8s.node.name" - # - "service.name" - # application_name: "{{.Values.global.defaultApplicationName }}" - # subsystem_name: "{{.Values.global.defaultSubsystemName }}" - processors: - k8sattributes: - filter: - node_from_env_var: KUBE_NODE_NAME - extract: - metadata: - - "k8s.namespace.name" - - "k8s.deployment.name" - - "k8s.statefulset.name" - - "k8s.daemonset.name" - - "k8s.cronjob.name" - - "k8s.job.name" - - "k8s.pod.name" - - "k8s.node.name" - memory_limiter: null # Will get the k8s resource limits - groupbytrace: - wait_duration: 1s - groupbyattrs: - keys: - - host.name - resourcedetection/env: - detectors: ["system","env"] - timeout: 5s - override: false - # spanmetrics: - # metrics_exporter: coralogix - # dimensions: - # - name: "k8s.deployment.name" - # - name: "k8s.statefulset.name" - # - name: "k8s.daemonset.name" - # - name: "k8s.cronjob.name" - # - name: "k8s.job.name" - # - name: "k8s.container.name" - # - name: "k8s.node.name" - # - name: "k8s.namespace.name" - receivers: - otlp: - protocols: - grpc: - endpoint: ${MY_POD_IP}:4317 - http: - endpoint: ${MY_POD_IP}:4318 - zipkin: - endpoint: ${MY_POD_IP}:9411 - jaeger: - protocols: - grpc: - endpoint: ${MY_POD_IP}:14250 - thrift_http: - endpoint: ${MY_POD_IP}:14268 - thrift_compact: - endpoint: ${MY_POD_IP}:6831 - thrift_binary: - endpoint: ${MY_POD_IP}:6832 - prometheus: - config: - scrape_configs: - - job_name: opentelemetry-collector - scrape_interval: 30s - static_configs: - - targets: - - ${MY_POD_IP}:8888 - service: - extensions: - # - zpages - # - pprof - - health_check -# - memory_ballast - telemetry: - metrics: - address: ${env:MY_POD_IP}:8888 - pipelines: - traces: - exporters: - - loadbalancing - processors: - - memory_limiter - - batch - receivers: - - otlp - metrics: null - logs: null -tolerations: - - operator: Exists - -resources: - requests: - cpu: 350m - memory: 350Mi - limits: - cpu: 1 - memory: 1G - -ports: - jaeger-binary: - enabled: true - containerPort: 6832 - servicePort: 6832 - hostPort: 6832 - protocol: TCP - # In order to enable podMonitor, following part must be enabled in order to expose the required port: - # metrics: - # enabled: true - -# podMonitor: -# enabled: true - -# prometheusRule: -# enabled: true -# defaultRules: -# enabled: true - -# Annotations to be added to pod -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external - - key: node_pool - operator: NotIn - values: - - np-central-default-prd-ase1 \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index 12da8c5..0000000 --- a/helm-overrides/k8s-central-prd-ase1/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,119 +0,0 @@ -fullnameOverride: opentelemetry-central-prd - -mode: daemonset - -priorityClassName: "system-node-critical" - -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmstorage - - megaduo - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external - - key: node_pool - operator: NotIn - values: - - np-central-default-prd-ase1 - -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 100m - memory: 124Mi - limits: - cpu: 200m - memory: 256Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "central" - team: "sre" - service: "opentelemetry-central-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index 579c955..0000000 --- a/helm-overrides/k8s-central-prd-ase1/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,497 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-admin-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - - -# Extra labels to be added to node exporter pods -podLabels: - bu: "central" - team: "central-sre" - service: "node-exporter-central-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-central-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index a1002e7..0000000 --- a/helm-overrides/k8s-central-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-central-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-central-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "central" - team: "central-sre" - service: "stackdriver-exporter-central-prd" - env: "prd" - priority: "p0" - type: "exporter" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-central-prd-0622" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,compute.googleapis.com/instance,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime,loadbalancing.googleapis.com/https' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: - - 'loadbalancing.googleapis.com/https:resource.labels.url_map_name=one_of("ext-lb-prd-cntr-xcntr-edge-guard-url-map","ext-lb-prd-cntr-xcntr-edge-guard-https-redirect","int-lb-prd-cntr-xcntr-edge-guard-url-map")' - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - dedicated: "vmselect-mds" - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-cntr-cnsre-stackdriver-prd@meesho-central-prd-0622.iam.gserviceaccount.com - } - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/k8s-central-prd-ase1/telegraf-operator/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/telegraf-operator/custom-values.yaml deleted file mode 100644 index 5df85b3..0000000 --- a/helm-overrides/k8s-central-prd-ase1/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,150 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "central" - team: "central-sre" - service: "telegraf-operator-central-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - dedicated: "vmselect-mds" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false diff --git a/helm-overrides/k8s-central-prd-ase1/temporal/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/temporal/custom-values.yaml deleted file mode 100644 index bc8347b..0000000 --- a/helm-overrides/k8s-central-prd-ase1/temporal/custom-values.yaml +++ /dev/null @@ -1,157 +0,0 @@ -fullnameOverride: "prd-central-shared-temporal" - -externalSecret: - enabled: true - path: prd/cntr/devop/shared-temporal - refreshInterval: "1m" - secretStoreRef: - kind: ClusterSecretStore - name: vault-backend - -server: - nodeSelector: - dedicated: central-devops - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - resources: - requests: - cpu: 500m - memory: 1024Mi - limits: - cpu: 500m - memory: 1024Mi - config: - namespaces: - create: true - namespace: - - name: default - retention: 3d - - name: cicd-service - retention: 3d - - name: central-rollout-service - retention: 3d - persistence: - default: - driver: "sql" - sql: - driver: "mysql8" - host: 10.147.2.79 - port: 3306 - database: temporal - user: temporal-user - existingSecret: prd-central-shared-temporal-secret - maxConns: 20 - maxIdleConns: 20 - maxConnLifetime: "1h" - visibility: - driver: "sql" - visibilityStore: es-visibility - sql: - driver: "mysql8" - host: 10.147.2.79 - port: 3306 - database: temporal_visibility - user: temporal-user - existingSecret: prd-central-shared-temporal-secret - maxConns: 20 - maxIdleConns: 20 - maxConnLifetime: "1h" - datastores: - es-visibility: # Define the Elasticsearch datastore connection information under the `es-visibility` key - elasticsearch: - version: "v7" - url: - scheme: "http" - host: "elasticsearch-master-headless:9200" - indices: - visibility: temporal_visibility_v1_dev - -admintools: - nodeSelector: - dedicated: central-devops - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 100m - memory: 128Mi - -web: - image: - repository: temporalio/ui - tag: 2.39.0 - pullPolicy: IfNotPresent - ingress: - enabled: true - className: contour-internal-1 - annotations: {} - hosts: - - "temporal.prd.meesho.int" - tls: [] - nodeSelector: - dedicated: central-devops - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - resources: - requests: - cpu: 1000m - memory: 1024Mi - limits: - cpu: 1000m - memory: 1024Mi - -elasticsearch: - enabled: true - replicas: 3 - persistence: - enabled: true - volumeClaimTemplate: - accessModes: ["ReadWriteOnce"] - storageClassName: premium-rwo - resources: - requests: - storage: 50Gi - imageTag: 7.17.3 - host: elasticsearch-master-headless - scheme: http - port: 9200 - version: "v7" - logLevel: "error" - username: "" - password: "" - visibilityIndex: "temporal_visibility_v1_dev" - nodeSelector: - dedicated: central-devops - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - resources: - requests: - cpu: 1 - memory: 6Gi - limits: - cpu: 1 - memory: 6Gi - -prometheus: - enabled: false -grafana: - enabled: false -cassandra: - enabled: false -mysql: - enabled: true \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml deleted file mode 100644 index 4979c4f..0000000 --- a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml +++ /dev/null @@ -1,296 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-central-prd-dr -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-central-prd-dr-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent-dr - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-cntr-cnsre-vmagent-prd@meesho-central-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - - https://vminsert-infra-prd-dr.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "central" - team: "central-sre" - service: "vmagent-central-prd" - env: "prd" - priority: "p0" - type: "vmagent-dr" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-central-prd-dr"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-central-prd-dr.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent-dr" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml deleted file mode 100644 index 224a20b..0000000 --- a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml +++ /dev/null @@ -1,272 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 1 - -fullnameOverride: vmagent-central-prd-fb -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-central-prd-fb-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - -# Remove topology spread constraints -topologySpreadConstraints: [] - -# Modify deployment strategy -deployment: - enabled: true - -# Remove HPA -horizontalPodAutoscaler: - enabled: false - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-infr-sre-vmagent-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -# Remove pod disruption budget -podDisruptionBudget: - enabled: false - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-central.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vmsinglenode-infra-prd.victoriametrics.svc.clusterset.local:8428/api/v1/write - - https://vmsinglenode-infra-prd.meeshogcp.in/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "central-fb" - team: "sre" - service: "vmagent-central-prd-fb" - env: "prd" - priority: "p0" - type: "vmagent-fb" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-central-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-central-prd-fb.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -# Modify resources -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - requests: - cpu: 2 - memory: 2Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmselect-mds" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-agent/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index 8c335ab..0000000 --- a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,296 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 3 - -fullnameOverride: vmagent-central-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-central-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-cntr-sre-vmagnt-prd-mds@meesho-central-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-central.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-central-prd.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "central" - team: "central-sre" - service: "vmagent-central-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-central-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-central-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 24 - memory: 40Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent-mds" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-mds" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml deleted file mode 100644 index dd0dff0..0000000 --- a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert-dr - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-central-prd-dr - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/central/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-central-prd-dr.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-central-prd-dr" - env: "prd" - priority: "p0" - type: "vmalert-dr" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - dedicated: "vmagent-dr" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert-secured-stateful/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert-secured-stateful/custom-values.yaml deleted file mode 100644 index d650dee..0000000 --- a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert-secured-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-stateful-secured-central-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-central-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-central-prd.victoriametrics.svc.cluster.local:8480/insert/200/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-central-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/sensitive/central/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: false - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-central-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "central" - team: "sre" - service: "vmalert-stateful-secured-central-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-secured-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert-secured/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert-secured/custom-values.yaml deleted file mode 100644 index 3a94103..0000000 --- a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert-secured/custom-values.yaml +++ /dev/null @@ -1,305 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-secured-central-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-central-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-central-prd.victoriametrics.svc.cluster.local:8480/insert/200/prometheus" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-central-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config1/sensitive/central/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-secured-central-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 1 - memory: 2Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-secured-central-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml deleted file mode 100644 index 8f09ef0..0000000 --- a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-central-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-central-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-central-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-central-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/central/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-central-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "central" - team: "sre" - service: "vmalert-central-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert/custom-values.yaml deleted file mode 100644 index 683e4a5..0000000 --- a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-alert/custom-values.yaml +++ /dev/null @@ -1,305 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-central-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-central-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-central-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-central-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config1/central/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-central-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-central-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-insert/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index 7a0b93d..0000000 --- a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,234 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-central-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-central-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 55 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "central" - team: "central-sre" - service: "vminsert-central-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 15 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 4 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vminsert-mds" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 3 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 5 - memory: 10Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-central-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-select/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index 7895794..0000000 --- a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,297 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-central-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - extraVMSelects: - # - --storageNode=http://vmselect-central-ase1c-prd.meeshogcp.in:8401 - # - --storageNode=http://vmselect-central-ase1c-prd.meeshogcp.in:8401 https endpoint - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-central-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - clusternative.maxConcurrentRequests: "256" - dedup.minScrapeInterval: 60s - search.maxConcurrentRequests: "96" - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 180s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "central" - team: "central-sre" - service: "vmselect-central-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 5 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "vmselect-mds" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 10 - memory: 10Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: contour-internal-0 - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-central.prd.meesho.int - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-storage/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index 8ea89f9..0000000 --- a/helm-overrides/k8s-central-prd-ase1/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,316 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-central-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage-sale-24aug" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage-sale-24aug" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 500Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "central" - team: "central-sre" - service: "vmstorage-central-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 10 - memory: 150Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.93.7-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-central-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-central-prd-ase1/victoriametrics-agent/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index 8b7eebd..0000000 --- a/helm-overrides/k8s-central-prd-ase1/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,481 +0,0 @@ -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "central-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -# -- Replica count -replicaCount: 2 - -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - -# -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables -deployment: - spec: - minReadySeconds: 1200 - progressDeadlineSeconds: 1800 - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - -image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - # -- Image tag, set to `Chart.AppVersion` by default - tag: v1.133.0 # rewrites Chart.AppVersion - # -- Variant of the image to use. - # e.g. enterprise, scratch - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - -# -- Image pull secrets -imagePullSecrets: [] - -# -- Add additional DNS entries to pods hosts file. Check [official documentation](https://kubernetes.io/docs/tasks/network/customize-hosts-file-for-pods/) -hostAliases: [] -# - ip: 192.168.1.1 -# hostNames: -# - test.example.com -# - another.example.net - -# -- Override chart name -nameOverride: "" - -# -- Override resources fullname -fullnameOverride: "vm-agent-central-prd" - -# -- Container working directory -containerWorkingDir: "/" - -rbac: - # -- Enables Role/RoleBinding creation - create: true - - # -- Role/RoleBinding annotations - annotations: {} - - # -- Role/RoleBinding labels - extraLabels: {} - - # -- If true and `rbac.enabled`, will deploy a Role/RoleBinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - - # -- additional rules for a role - extraRules: [] - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - # -- Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-cntr-sre-vmagnt-prd-mds@meesho-central-prd-0622.iam.gserviceaccount.com - } - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # -- mount API token to pod directly - automountToken: true - -# -- See `kubectl explain poddisruptionbudget.spec` for more or check [official documentation](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# -- Generates `remoteWrite.*` flags and config maps with value content for values, that are of type list of map. -# Each item should contain `url` param to pass validation. -remoteWrite: - - url: http://vm-insert-central-prd.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write - disableOnDiskQueue: false - dropSamplesOnOverload: false - - url: http://prd-census-server.prd-census-server.svc.cluster.local/api/v1/write - disableOnDiskQueue: true - dropSamplesOnOverload: true - -# -- VMAgent extra command line arguments -extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8429 - promscrape.config.strictParse: false # default is true - promscrape.maxScrapeSize: 1000000000 # default is 167772160 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - # promscrape.suppressScrapeErrors: true # default is false - - # Uncomment and specify the port if you want to support any of the protocols: - # https://docs.victoriametrics.com/victoriametrics/vmagent/#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for more details. -env: - - name: GOGC - value: "200" - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth-secret - # key: password - -# -- Specify alternative source for env variables -envFrom: - [] - #- configMapRef: - # name: special-config - -# -- Extra labels for Deployment and Statefulset -extraLabels: - bu: "central" - team: "central-sre" - service: "vm-agent-central-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra labels for Pods only -podLabels: - bu: "central" - team: "central-sre" - service: "vm-agent-central-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra selector labels common for pod and service -selectorLabels: {} - -# -- Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# -- Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# -- Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -# -- Extra containers to run in a pod with vmagent -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -# -- Init containers for vmagent -initContainers: - [] - # - name: example - # image: example-image - -# -- Security context to be added to pod -podSecurityContext: - enabled: true - # fsGroup: 2000 - -# -- Security context to be added to pod's containers -securityContext: - enabled: true - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - # -- Enable agent service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - extraLabels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) for details - externalIPs: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8429 - # -- Target port - targetPort: http - # nodePort: 30000 - # -- Service type - type: ClusterIP - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - -ingress: - # -- Enable deployment of ingress for agent - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-agent-central-prd.meeshogcp.in - path: - - / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - - # -- Ingress controller class name - ingressClassName: "nginx-internal" - - # -- Ingress path type - pathType: Prefix - -# -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) -resources: - requests: - cpu: 27 - memory: 50Gi - -# -- Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) -nodeSelector: - dedicated: "vmagent-mds" - -# -- Node tolerations for server scheduling to nodes with taints. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-mds" - effect: "NoSchedule" - -# -- Pod topologySpreadConstraints -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# -- Pod affinity -affinity: {} - -# -- VMAgent [scraping configuration](https://docs.victoriametrics.com/victoriametrics/vmagent/#how-to-collect-metrics-in-prometheus-format) -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-central-prd" - -# -- Priority class to be assigned to the pod(s) -priorityClassName: "" - -# -- Enable the host network -hostNetwork: false - -serviceMonitor: - # -- Enable deployment of Service Monitor for server component. This is Prometheus operator object - enabled: false - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Service Monitor relabelings - relabelings: [] - # -- Basic auth params for Service Monitor - basicAuth: {} - # -- Service Monitor metricRelabelings - metricRelabelings: [] - # -- Service Monitor targetPort - targetPort: http - # interval: 15s - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - -# -- Empty dir configuration for a case, when persistence is disabled -emptyDir: {} - -persistentVolume: - # -- Create/use Persistent Volume Claim for server component. Empty dir if false - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Size of the volume. Should be calculated based on the logs you send and retention policy you set. - size: 10Gi - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume additional labels - extraLabels: {} - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Horizontal Pod Autoscaling. -# Note that it is not intended to be used for vmagents which perform scraping. -# In order to scale scraping vmagents check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) -horizontalPodAutoscaling: - # -- Use HPA for vmagent - enabled: false - # -- Maximum replicas for HPA to use to to scale vmagent - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale vmagent - minReplicas: 1 - # -- Metric for HPA to use to scale vmagent - metrics: [] - -# -- VMAgent scrape configuration -config: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -probe: - # -- Readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - # -- Liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - # -- Startup probe - startup: {} - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -allowedMetricsEndpoints: - - /metrics - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-central-prd-ase1/victoriametrics-insert/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/victoriametrics-insert/custom-values.yaml deleted file mode 100644 index 797511b..0000000 --- a/helm-overrides/k8s-central-prd-ase1/victoriametrics-insert/custom-values.yaml +++ /dev/null @@ -1,411 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: false - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- IDs of vmstorage nodes to exclude from writing - excludeStorageIDs: [] - # -- Override default `app` label name - name: vminsert - # -- VMInsert strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - # override Chart.AppVersion - tag: v1.107.0-cluster - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMInsert http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-central-prd - # -- Extra command line arguments for vminsert component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8480 - maxLabelsPerTimeseries: 55 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-central-prd-0.vm-storage-central-prd.victoriametrics.svc:8400" - - "vm-storage-central-prd-1.vm-storage-central-prd.victoriametrics.svc:8400" - - "vm-storage-central-prd-2.vm-storage-central-prd.victoriametrics.svc:8400" - - "vm-storage-central-prd-3.vm-storage-central-prd.victoriametrics.svc:8400" - - "vm-storage-central-prd-4.vm-storage-central-prd.victoriametrics.svc:8400" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "central" - team: "central-sre" - service: "vm-insert-central-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Pod's additional labels - podLabels: - bu: "central" - team: "central-sre" - service: "vm-insert-central-prd" - env: "prd" - priority: "p0" - type: "vminsert" - terminationGracePeriodSeconds: 30 - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - - # -- Readiness & Liveness probes - probe: - # -- VMInsert readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert startup probe - startup: {} - - # -- Relabel configuration - relabel: - enabled: false - config: [] - # -- Use existing configmap if specified - # otherwise .config values will be used. Relabel config **should** reside under `relabel.yml` key - configMap: "" - - # Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 15 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 4 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vminsert - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vminsert - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - dedicated: "vminsert-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 6 - memory: 10Gi - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - service: - # -- Create VMInsert service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here]( https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) for details - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-insert-central-prd.meeshogcp.in - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - - # -- Ingress controller class name - ingressClassName: nginx-internal - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for insert component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/insert" .Values.vminsert }}' - - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vminsert component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-central-prd-ase1/victoriametrics-select/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/victoriametrics-select/custom-values.yaml deleted file mode 100644 index 23a16ae..0000000 --- a/helm-overrides/k8s-central-prd-ase1/victoriametrics-select/custom-values.yaml +++ /dev/null @@ -1,449 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - splitService: true - externalService: - enabled: true - name: vmselect-central-prd-proxy - # -- Override default `app` label name - - clusternativeService: - enabled: false - - name: "" - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # override Chart.AppVersion - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMSelect http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-central-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - # -- Extra command line arguments for vmselect component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8481 - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 180s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 96 - clusternative.maxConcurrentRequests: 256 - storageNode: - - "vm-storage-central-prd-0.vm-storage-central-prd.victoriametrics.svc:8401" - - "vm-storage-central-prd-1.vm-storage-central-prd.victoriametrics.svc:8401" - - "vm-storage-central-prd-2.vm-storage-central-prd.victoriametrics.svc:8401" - - "vm-storage-central-prd-3.vm-storage-central-prd.victoriametrics.svc:8401" - - "vm-storage-central-prd-4.vm-storage-central-prd.victoriametrics.svc:8401" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "central" - team: "central-sre" - service: "vm-select-central-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # -- Pod's additional labels - podLabels: - bu: "central" - team: "central-sre" - service: "vm-select-central-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Readiness & Liveness probes - probe: - # -- VMSelect readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 5 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vmselect - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vmselect - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - dedicated: "vmselect-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 18 - memory: 30Gi - - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Create VMSelect service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: false - - # -- Ingress annotations - annotations: {} - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-select-central.prd.meesho.int - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # -- Ingress controller class name - ingressClassName: contour-internal-0 - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for select component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/select" .Values.vmselect }}' - - # -- vmselect mode: deployment, daemonSet - mode: deployment - - # -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables - deployment: - spec: - # -- VMSelect strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - - # -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables - statefulSet: - enabled: false - spec: - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume extra labels - extraLabels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmselect component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: false - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - diff --git a/helm-overrides/k8s-central-prd-ase1/victoriametrics-storage/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1/victoriametrics-storage/custom-values.yaml deleted file mode 100644 index 8355475..0000000 --- a/helm-overrides/k8s-central-prd-ase1/victoriametrics-storage/custom-values.yaml +++ /dev/null @@ -1,363 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - enabled: false - -vminsert: - enabled: false - -vmauth: - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- Override default `app` label name - name: vmstorage - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Variant of the image to use. e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMStorage http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-central-prd - - # -- Additional environment variables (ex.: secret tokens, flags). Check https://docs.victoriametrics.com/victoriametrics/#environment-variables for details - env: [] - # -- Specify alternative source for env variables - envFrom: [] - # -- Data retention period. Possible units character: h(ours), d(ays), w(eeks), y(ears), if no unit character specified - month. The minimum retention period is 24h. - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - httpListenAddr: :8482 - loggerTimezone: "Asia/Kolkata" - dedup.minScrapeInterval: 60s - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage-n4d" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage-n4d" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - # -- Use an alternate scheduler, e.g. "stork". Check https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ for details - # schedulerName: - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Override Persistent Volume Claim name - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are https://kubernetes.io/docs/concepts/storage/persistent-volumes/ - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume extra labels - extraLabels: {} - # -- Storage class name. Will be empty if not set - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume - size: 1200Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "central" - team: "central-sre" - service: "vm-storage-central-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - # -- Pod's additional labels - podLabels: - bu: "central" - team: "central-sre" - service: "vm-storage-central-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Details are https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 25 - memory: 225Gi - - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: - enabled: false - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - podSecurityContext: - enabled: false - service: - enabled: true - # -- Service annotations - annotations: {} - # -- Service ClusterIP - clusterIP: None - # -- Service type - type: ClusterIP - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmselect - vmselectPort: 8401 - # -- Extra service ports - extraPorts: [] - # -- Health check node port for a service. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilyPolicy: "" - # -- List of service IP families. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilies: [] - # -- Traffic Distribution. Check https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution - trafficDistribution: "" - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - minReadySeconds: 5 - # -- Readiness probes - probe: - # -- VMStorage readiness probe - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMStorage startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmstorage component - enabled: false - # -- Maximum replicas for HPA to use to to scale the vmstorage component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmstorage component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmstorage component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: - scaleDown: - selectPolicy: Disabled - - vmbackupmanager: - # -- Enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-central-prd-ase1c/conntrack-adjuster/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 7b64b15..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - # Maximum number of conntrack entries - max: 2097152 - # Hash size for conntrack - hashsize: 524288 - # Sleep interval between adjustments (seconds) - sleepInterval: 30 - -# Enable additional matchExpressions -# addExtraMatchExpressions: true - -# Additional matchExpressions to append -# additionalMatchExpressions: -# - key: node-role -# operator: In -# values: -# - "worker" -# - "ingress" -# - key: environment -# operator: In -# values: -# - "production" -# - "staging" \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1c/contour-cert-checker/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/contour-cert-checker/custom-values.yaml deleted file mode 100644 index 2d88f85..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "0 12 * * *" - args: ["--cluster=k8s-central-prd-ase1c"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - dedicated: central-devops - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1c/contour-external-1/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/contour-external-1/custom-values.yaml deleted file mode 100644 index d0b73ac..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/contour-external-1/custom-values.yaml +++ /dev/null @@ -1,86 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external-1 - nodeSelector: - dedicated: contour-external-1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external-1 - nodeSelector: - dedicated: contour-external-1 - logLevel: error - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "30" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-1-central-c-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-central-prd-ase1c/contour-external/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/contour-external/custom-values.yaml deleted file mode 100644 index bd13f85..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/contour-external/custom-values.yaml +++ /dev/null @@ -1,86 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external - nodeSelector: - dedicated: contour-external - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external - nodeSelector: - dedicated: contour-external - logLevel: error - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "30" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-central-prd-c"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-central-prd-ase1c/contour-internal-0/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/contour-internal-0/custom-values.yaml deleted file mode 100644 index 3f17eff..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,90 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - image: - tag: 1.28.0-debian-11-r8 - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-central-c-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-central-prd-ase1c/contour-internal-1/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/contour-internal-1/custom-values.yaml deleted file mode 100644 index 9d38e77..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,91 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 360s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - image: - tag: 1.28.0-debian-11-r8 - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - logLevel: error - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int1-central-c-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-central-prd-ase1c/contour-internal-intra-0/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index d7e2ba6..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,88 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - image: - tag: 1.28.0-debian-11-r8 - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-central-prd-ase1c/contour-internal-intra-1/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index 5e29936..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,89 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 360s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: central - team: central-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - image: - tag: 1.28.0-debian-11-r8 - enabled: true - podLabels: - bu: central - team: central-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - logLevel: error - autoscaling: - enabled: true - minReplicas: 5 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-central-prd-ase1c/coredns/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/coredns/custom-values.yaml deleted file mode 100644 index b9cb879..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/coredns/custom-values.yaml +++ /dev/null @@ -1,31 +0,0 @@ -replicaCount: 16 - -labels: - bu: central - team: central-devops - env: prd - -clusterIP: 10.207.16.2 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - -nodeSelector: - dedicated: central-devops - kubernetes.io/os: linux - - -overwriteRegion: "c" -communicationType: "" \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1c/external-secrets/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/external-secrets/custom-values.yaml deleted file mode 100644 index 61baf20..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/external-secrets/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Additional service annotations - annotations: {} - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - nodeSelector: - dedicated: central-devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - nodeSelector: - dedicated: central-devops - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - nodeSelector: - dedicated: central-devops diff --git a/helm-overrides/k8s-central-prd-ase1c/flagger/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/flagger/custom-values.yaml deleted file mode 100644 index 58317fc..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/flagger/custom-values.yaml +++ /dev/null @@ -1,68 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: hpa-changes-26 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - dedicated: central-devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: central-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: central-devops - bu: infra - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-central-rollout-service.prd-central-rollout-service.svc.cluster.local" - - -rolloutService: - enabled: true - apps: - prd-devops-node-app-consumers: true - prd-devops-node-app: true - prd-devops-golang-app: true - prd-devops-golang-app-test: true - prd-devops-golang-app-crons: true diff --git a/helm-overrides/k8s-central-prd-ase1c/fluentd/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/fluentd/custom-values.yaml deleted file mode 100644 index 291ba16..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/fluentd/custom-values.yaml +++ /dev/null @@ -1,720 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-cntr-cnsre-fluentd-prd@meesho-central-ase1c-prd-0625.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: central - team: sre - type: fluentd - service: fluentd-central-prd - priority: p0 - env: prd - zone_extended: ase1c - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: central - team: sre - type: fluentd - service: fluentd-central-prd - priority: p0 - env: prd - zone_extended: ase1c - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: [] -# - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - # - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-central-prd-ase1c/ingress-nginx/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/ingress-nginx/custom-values.yaml deleted file mode 100644 index 715fdaf..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/ingress-nginx/custom-values.yaml +++ /dev/null @@ -1,32 +0,0 @@ -ingress-nginx: - controller: - config: - proxy-body-size: "50g" - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - resources: - requests: - cpu: 200m - memory: 512Mi - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 30 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-central-ase1c-prd"}}}' - nodeSelector: - dedicated: devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-central-prd-ase1c/keda/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/keda/custom-values.yaml deleted file mode 100644 index 23eb546..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/keda/custom-values.yaml +++ /dev/null @@ -1,35 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: central-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - podLabels: - bu: "central" - team: "central-devops" - metricsAdapter: - bu: "central" - team: "central-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 100m - memory: 200Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 50m - memory: 150Mi - requests: - cpu: 10m - memory: 80Mi \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1c/kube-dns/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/kube-dns/custom-values.yaml deleted file mode 100644 index cea794e..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.207.16.2"],"prd.mrouter.int.svc.cluster.local":["10.207.16.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.207.16.2"]} diff --git a/helm-overrides/k8s-central-prd-ase1c/kube-events/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/kube-events/custom-values.yaml deleted file mode 100644 index 0dea214..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-central-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - dedicated: central-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: central-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: central-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: central-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: central-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: central-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-central-prd-ase1c/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 97c1621..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-central-ase1c-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-central-ase1c-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "central" - team: "central-sre" - service: "kube-state-metrics-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "central-devops" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "central-devops" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 23m - memory: 169Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-central-prd-ase1c/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index 919a0ff..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,119 +0,0 @@ -fullnameOverride: opentelemetry-central-ase1c-prd - -mode: daemonset - -priorityClassName: "system-node-critical" - -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-central-prd-ase1c-headless.opentelemetry.svc.cluster.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external - - key: node_pool - operator: NotIn - values: - - np-central-default-prd-ase1 - -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 100m - memory: 124Mi - limits: - cpu: 200m - memory: 256Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "central" - team: "sre" - service: "opentelemetry-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - zone_extended: "ase1c" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1c/opentelemetry-deployment/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/opentelemetry-deployment/custom-values.yaml deleted file mode 100644 index 297aee9..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/opentelemetry-deployment/custom-values.yaml +++ /dev/null @@ -1,111 +0,0 @@ -config: - - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - max_recv_msg_size_mib: 50 - - processors: - batch: - send_batch_size: 256 - timeout: 200ms - send_batch_max_size: 512 - filter/drop-noisy-services: - error_mode: ignore - traces: - span: - - IsMatch(resource.attributes["service.name"], ".*consumer.*|.*scheduler.*|.*cron.*|.*worker.*|.*inhouse-ingestion.*|.*.messaging-api-internal*|.*cis.*") - tail_sampling: - decision_wait: 10s # avg latency across services - num_traces: 25000000 # expected_new_traces_per_sec * decision_wait + some buffer - expected_new_traces_per_sec: 1500000 # combined span rate across all business units - decision_cache: - sampled_cache_size: 6000000 # sampling rate * num_traces + some buffer - policies: - [ - { - name: errors-policy, - type: status_code, - status_code: {status_codes: [ERROR]} - }, - # { - # name: latency-policy, - # type: latency, - # latency: {threshold_ms: 1000} - # }, - { - name: probablistic-policy, - type: probabilistic, - probabilistic: {sampling_percentage: 1} - } - ] - - exporters: - otlp/elastic: - endpoint: "946ece6f7b344005aa1e5c273b1a886f.apm.psc.asia-southeast1.gcp.elastic-cloud.com:443" - timeout: 15s - sending_queue: - num_consumers: 50 - queue_size: 10000 - headers: - Authorization: "Bearer ehllv9fQQ7GGMOjE79" - - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - traces: - receivers: [otlp] - processors: [filter/drop-noisy-services, tail_sampling, batch] - exporters: [otlp/elastic] - metrics: null - logs: null - -fullnameOverride: "opentelemetry-central-ase1c-prd" - -mode: "deployment" - -podAnnotations: - otel.io/path: '/metrics' - otel.io/port: '8888' - otel.io/scrape: 'true' - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - tag: "0.114.0" - -nodeSelector: - dedicated: "opentelemetry-std" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "opentelemetry-std" - effect: "NoSchedule" - -resources: - requests: - cpu: '28' - memory: 115Gi - limits: - cpu: '28' - memory: 115Gi - -autoscaling: - enabled: true - minReplicas: 8 - maxReplicas: 50 - targetCPUUtilizationPercentage: 75 - targetMemoryUtilizationPercentage: 75 \ No newline at end of file diff --git a/helm-overrides/k8s-central-prd-ase1c/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index 12066c1..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,498 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-admin-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - - -# Extra labels to be added to node exporter pods -podLabels: - bu: "central" - team: "central-sre" - service: "node-exporter-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - zone_extended: "ase1c" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-central-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index 7b2c1b0..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-central-ase1c-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-central-ase1c-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "central" - team: "central-sre" - service: "stackdriver-exporter-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - zone_extended: "ase1c" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-central-ase1c-prd-0225" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,compute.googleapis.com/instance,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - dedicated: "vmselect" - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-cntr-cnsre-stackdriver-prd@meesho-central-ase1c-prd-0622.iam.gserviceaccount.com - } - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/k8s-central-prd-ase1c/telegraf-operator/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/telegraf-operator/custom-values.yaml deleted file mode 100644 index b9ae2ea..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,151 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "central" - team: "central-sre" - service: "telegraf-operator-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - zone_extended: "ase1c" - -nodeSelector: - dedicated: "vmselect" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false diff --git a/helm-overrides/k8s-central-prd-ase1c/victoria-metrics-agent/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index dd50588..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,298 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-central-ase1c-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-central-ase1c-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-common-np-cntr-prd@meesho-central-ase1c-prd-0625.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - http://vminsert-central-ase1c-prd.cntr-c.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-central-ase1c-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write - # - http://vminsert-central-ase1c-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "central" - team: "central-sre" - service: "vmagent-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmagent" - zone_extended: "ase1c" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-central-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-central-ase1c-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: '4' - memory: 4Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-central-prd-ase1c/victoria-metrics-insert/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index 5845c5f..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,235 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-central-ase1c-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-central-ase1c-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "central" - team: "central-sre" - service: "vminsert-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "vminsert" - zone_extended: "ase1c" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 1 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vminsert" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 3 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: '2' - memory: 2Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-central-ase1c-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-central-prd-ase1c/victoria-metrics-select/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index 6b536aa..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,301 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-central-ase1c-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-central-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - clusternative.maxConcurrentRequests: "256" - dedup.minScrapeInterval: 60s - search.maxConcurrentRequests: "96" - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 180s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "central" - team: "central-sre" - service: "vmselect-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmselect" - zone_extended: "ase1c" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 15 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 1 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "vmselect" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 1 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 5 - memory: 3Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-central-ase1c-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - clusternativeService: - enabled: true - port: 8401 - sessionAffinity: None - ipFamilyPolicy: SingleStack - - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-central-prd-ase1c/victoria-metrics-storage/custom-values.yaml b/helm-overrides/k8s-central-prd-ase1c/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index e522462..0000000 --- a/helm-overrides/k8s-central-prd-ase1c/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,317 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-central-ase1c-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 250Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "central" - team: "central-sre" - service: "vmstorage-central-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - zone_extended: "ase1c" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 4 - memory: 53Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.93.7-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-central-ase1c-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dataengg-prd-ase1/README.md b/helm-overrides/k8s-dataengg-prd-ase1/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/k8s-dataengg-prd-ase1/alloy/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/alloy/custom-values.yaml deleted file mode 100644 index 8d18767..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/alloy/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -fullnameOverride: "alloy-dataengg-prd" - -alloy: - configMap: - configFile: dataengg.alloy - - clustering: - enabled: true - - extraPorts: - - name: "otlp-grpc" - port: 4317 - targetPort: 4317 - protocol: "TCP" - - name: "otlp-http" - port: 4318 - targetPort: 4318 - protocol: "TCP" - - resources: - requests: - cpu: 7 - memory: 55Gi - -configReloader: - enabled: true - -serviceAccount: - annotations: { - iam.gke.io/gcp-service-account: sa-deng-sre-grafna-obs-stk-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - } - -controller: - # Must be one of 'daemonset', 'deployment', or 'statefulset'. - type: 'deployment' - - nodeSelector: - dedicated: "alloy" - tolerations: - - key: "dedicated" - operator: "Equal" - value: "alloy" - effect: "NoSchedule" - - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 50 - targetCPUUtilizationPercentage: 80 - targetMemoryUtilizationPercentage: 80 \ No newline at end of file diff --git a/helm-overrides/k8s-dataengg-prd-ase1/aurva-dataplane/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/aurva-dataplane/custom-values.yaml deleted file mode 100644 index 975e209..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/aurva-dataplane/custom-values.yaml +++ /dev/null @@ -1,675 +0,0 @@ -## Aurva Data Plane -## Ref: https://github.com/aurva-io/aurva-charts.git - -postgresql: - enabled: true - fullnameOverride: "aurva-dataplane-database" - volumePermissions: - ## @param volumePermissions.enabled Enable init container that changes the owner and group of the persistent volume - ## - enabled: true - global: - storageClass: pd-standard-retain - postgresql: - auth: - postgresPassword: "aurva" - database: "controller" - # Add toleration to make sure where this postgres db pod should reside (Applicable for production workloads): For more detail ref: https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/ - primary: - extendedConfiguration: | - max_connections = 300 - - #PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dataengg-devops - - # -- Select nodes to deploy which matches the following labels - nodeSelector: ##PLACEHOLDER## - dedicated: dataengg-devops - -# -- Provide a name in place of `aurva` -# namespaceOverride: aurva-dataplane - -########################################################## -# Global Configs -########################################################## -global: - aurva_controller: - enabled: true - aurva_fastdet: - enabled: false - aurva_pii_analyzer: - enabled: true - aurva_ocr: - enabled: true - aurva_collector: - enabled: true - - deploymentAnnotations: {} - - priorityClassName: "" - -########################################################## -# Aurva Controller -########################################################## -aurva_controller: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "dataengg" - team: "dataengg-devops" - service: "aurva-dataengg-prd" - env: "prd" - priority: "p0" - type: "aurva_controller" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 10 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - image: - # Image of the app container - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-controller/aurva-controller - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - - type: secret - name: aurva-controller-secrets - # - type: configmap - # name: proxy-datasource-config - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 2 - requests: - memory: 1Gi - cpu: 1 - - aurvaFastdet: - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-fastdet/aurva-fastdet - tag: "v2.30.13" - pullPolicy: IfNotPresent - envFrom: [] - env: [] - resources: - limits: - cpu: 0.5 - memory: 512Mi - requests: - cpu: 0.5 - memory: 512Mi - - nodeSelector: ##PLACEHOLDER## - dedicated: dataengg-devops - - # -- Taint tolerations for nodes - ##PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dataengg-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-controller-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - config: - #variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - COMMAND_URL: "command.aurva-prd.meeshogcp.in:80" - DEPLOYMENT_TYPE: "kubernetes" - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - FLUSHER_WORKER_POOL_SIZE: "1000" - FLUSHER_BATCH_SIZE: "30000" - UNIQUENESS_IDENTIFIER: "k8s-dataengg-prd-ase1" #Recommendation: should be equal to cluster name - PROVIDER_ACCOUNT_ID: "meesho-dataengg-prd-0622" # GCP PROJECT ID (not Number) - REGION: "asia-southeast1" #eg: asia-south1 - ENVIRONMENT: "prod" - OCR_ENABLED: "true" - MONITORING_ENABLED: "false" - HYBRID_ONLY_MODE: "false" - FORCE_TLS: "false" - FASTDET_FLAG : "true" - AADHAAR_ENHANCER: "0" - WORKSPACE_EVENT_TRACKING_ENABLED: "false" - ACCESS_IQ_ENABLED: "true" - GRPC_ENFORCE_ALPN_ENABLED: "false" - ENABLE_FASTDET: "true" - FASTDET_MAX_BATCH_SIZE: "100" - HEARTBEAT_INTERVAL: "5m" - #pii data - PII_BUCKET_NAME: "gcs-infra-devop-aurva-dataengg-prd" - PII_LOG_BUCKET_REGION: "asia-southeast1" - PII_LOG_CRON: "*/5 * * * *" - ENABLE_PII_LOG: "true" - PII_EVIDENCE_UPLOAD_MAX_BLOCKING_TASKS: "50000" - PII_EVIDENCE_UPLOAD_MAX_CONCURRENT_TASKS: "500" - PII_EVIDENCE_MAX_CACHE_WEIGHT: "100" - QUOTA_CLEANUP_CRON: "0 0 * * *" - ENABLE_PII_QUOTA: "true" - MAX_PII_EVIDENCES_PER_KEY_PER_WINDOW: "3" - PII_QUOTA_SYNC_CRON: "*/2 * * * *" - QUOTA_WINDOW_HOURS: "24" - #constants - SKIP_NAMESPACES: "contour-internal-1-dataengg-prd,contour-internal-0-dataengg-prd,contour-external-dataengg-prd" - CLOUD_PROVIDER: "gcp" - LOG_ENV: "production" - RDS_SCANNER_AVAILABILITY : "false" - REDSHIFT_SCANNER_AVAILABILITY : "false" - S3_SCANNER_AVAILABILITY : "false" - DYNAMO_SCANNER_AVAILABILITY: "false" - DOCDB_SCANNER_AVAILABILITY: "false" - OPENSEARCH_SCANNER_AVAILABILITY: "false" - CLOUDSQL_SCANNER_AVAILABILITY: "true" - BIGQUERY_SCANNER_AVAILABILITY: "true" - AWS_SNAPSHOT_SCANNER_AVAILABILITY: "false" - CLOUDSTORAGE_SCANNER_AVAILABILITY: "true" - KEYSPACES_SCANNER_AVAILABILITY: "false" - ALLOYDB_SCANNER_AVAILABILITY: "true" - BIGTABLE_SCANNER_AVAILABILITY: "true" - GCP_BACKUP_AVAILABILITY: "true" - EGRESS_MODE_ONLY: "false" - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - SCAN_UUID_ENABLED: "true" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-controller-sa - # -- Annotations applied to created service account - annotations: - iam.gke.io/gcp-service-account: sa-dataengg-prd-aurva-contr@meesho-dataengg-prd-0622.iam.gserviceaccount.com -# eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 15 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 - - -########################################################## -# Aurva OCR -########################################################## -aurva_ocr: - # -- Additional labels for aurva-controller - additionalLabels: - bu: "dataengg" - team: "dataengg-devops" - service: "aurva-dataengg-prd" - env: "prd" - priority: "p0" - type: "aurva_ocr" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 1 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-ocr/aurva-ocr - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-ocr: - type: secret - name: aurva-ocr-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 1 - requests: - memory: 2Gi - cpu: 1 - - # -- Select nodes to deploy which matches the following labels - - nodeSelector: ##PLACEHOLDER## - dedicated: dataengg-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dataengg-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-ocr-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - OCR_TIME_LIMIT: "1" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-dataengg-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-ocr-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # iam.gke.io/gcp-service-account: service-account@gcp.iam.gserviceaccount.com - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva Collector -########################################################## -aurva_collector: - - # -- Additional labels for aurva-analyzer - additionalLabels: - bu: "dataengg" - team: "dataengg-devops" - service: "aurva-dataengg-prd" - env: "prd" - priority: "p0" - type: "aurva_collector" - - - # -- Annotations on aurva-analyzer - annotations: {} -# "key": "value" - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-collector/aurva-collector - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-controller: - type: secret - name: aurva-collector-secrets - - resources: - limits: - cpu: 800m - memory: 800Mi - requests: - cpu: 200m - memory: 512Mi - - podSecurityContext: {} - - securityContext: - privileged: true - capabilities: - add: - # For kernel v5.8 and above we don't need CAP_SYS_ADMIN or CAP_SYS_RESOURCE - # we just need CAP_BPF and CAP_PERFMON. This has been tested on our EKS node - # which is on kernel v5.10.x - # When SSL Tracing is required we need CAP_SYS_ADMIN and CAP_SYS_PTRACE - # on top of the previous capabilities - # So finally these are the 4 possible combinations for capabilies - # 1. Newer Kernels without SSL - # - BPF - # - PERFMON - # 2. Newer Kernels with SSL - - SYS_ADMIN - - SYS_PTRACE - # 3. Older Kernels without SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # 4. Older Kernels with SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # - SYS_PTRACE - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - volumes: - - name: debugfs - mountPath: /sys/kernel/debug - hostPath: /sys/kernel/debug - - name: vmlinux - mountPath: /sys/kernel/btf/vmlinux - hostPath: /sys/kernel/btf/vmlinux - - name: procfs - mountPath: /host/proc - hostPath: /proc - - name: bpffs - mountPath: /sys/fs/bpf - hostPath: /sys/fs/bpf - - # -- Taint tolerations for nodes - tolerations: - # - effect: NoSchedule - # key: dedicated - # operator: Equal - # value: megatetra - - operator: Exists - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external - - alloy - - preprod-spot-16 - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-collector-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - # variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-dataengg-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - TRACE_INTERNAL_SVC: "true" - TRACE_INTERNAL_SVC_HTTP: "true" - INTERNAL_SVC_SAMPLE_INTERVAL: "10m" - LOGS_TTL: "1h" - TRACE_HTTP2: "true" - TRACE_SSL: "false" - TRACE_PSQL: "false" - TRACE_SQLSERVER: "false" - TRACE_MYSQL: "false" - TRACE_EGRESS: "true" - TRACE_GO_TLS: "false" - TRACE_ML_SERVICES: "false" - # constants - LOG_ENV: production - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - MONITORING_ENABLED: "false" - ENABLE_INGRESS_INFORMER: "false" - ENABLE_SERVICE_INFORMER: "false" - ENABLE_ISTIO_INFORMER: "false" - EXCLUDED_PII_REGEX_TYPES: "ip_address,us_bank_number,us_driver_license,us_itin,us_passport,us_routing,us_mbi,ssn" - AGGREGATOR_MAX_CONNECTIONS: "1000" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-collector-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva PII Analyzer -########################################################## -aurva_pii_analyzer: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "dataengg" - team: "dataengg-devops" - service: "aurva-dataengg-prd" - env: "prd" - priority: "p0" - type: "aurva_pii_analyzer" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 3 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - ##PLACEHOLDER## - nodeSelector: - dedicated: dataengg-devops - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-piianalyzer/aurva-piianalyzer - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-pii-analyzer: - type: secret - name: aurva-pii-analyzer-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 4Gi - cpu: 4 - requests: - memory: 2Gi - cpu: 2 - - nodeSelector: ##PLACEHOLDER## - dedicated: dataengg-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dataengg-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-pii-analyzer-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - SCHEDULER_TIME: "1" - SUPPORTED_REGION: "US" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-dataengg-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-pii-analyzer-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 3 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 diff --git a/helm-overrides/k8s-dataengg-prd-ase1/cert-manager/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/cert-manager/custom-values.yaml deleted file mode 100644 index 24756b4..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/cert-manager/custom-values.yaml +++ /dev/null @@ -1,129 +0,0 @@ -global: - logLevel: 2 - rbac: - create: true - priorityClassName: "high-priority" -installCRDs: false - -crds: - enabled: true - keep: true - -# Cert-manager Controller -replicaCount: 1 -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-controller - tag: v1.20.1 - pullPolicy: IfNotPresent - -nodeSelector: - dedicated: dataengg-devops - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "dataengg-devops" - effect: "NoSchedule" - -resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 500m - memory: 512Mi - -# Webhook Configuration -webhook: - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-webhook - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: dataengg-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "dataengg-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# CA Injector Configuration -cainjector: - enabled: true - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-cainjector - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: dataengg-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "dataengg-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# ACME Solver Configuration -acmesolver: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-acmesolver - tag: v1.20.1 - pullPolicy: IfNotPresent - -# Startup API Check -startupapicheck: - enabled: true - timeout: 1m - backoffLimit: 4 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-startupapicheck - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: dataengg-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "dataengg-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - -# Prometheus Monitoring -prometheus: - enabled: true - servicemonitor: - enabled: false - interval: 60s - scrapeTimeout: 30s - labels: - prometheus: cert-manager diff --git a/helm-overrides/k8s-dataengg-prd-ase1/conntrack-adjuster/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index b56f52c..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,19 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - max: 2097152 - hashsize: 524288 - sleepInterval: 30 - -nodeAffinity: - enabled: true - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: In - values: - - "contour-external" - - "contour-internal-0" - - "contour-internal-1" - - "contour-intra-1" diff --git a/helm-overrides/k8s-dataengg-prd-ase1/contour-ca-issuer/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/contour-ca-issuer/custom-values.yaml deleted file mode 100644 index 618b576..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/contour-ca-issuer/custom-values.yaml +++ /dev/null @@ -1,17 +0,0 @@ -# ClusterIssuer override for k8s-dataengg-prd-ase1 (prd dataengg cluster, ase1a zone). -# ClusterIssuer is a cluster-scoped resource — this file pins the issuer -# identity for this cluster so the name is auditable per-cluster. -# -# Must match the `issuerRef.name` in consuming Certificate CRs (see -# devops-helm-charts/2.0.0/templates/proxyless-grpc-cert.yaml). - -issuerName: contour-dataengg-prd-ca-issuer -rootCASecretName: contour-dataengg-ca - -externalSecret: - enabled: true - vaultPath: meesho/devops/contour/root-ca - refreshInterval: "0" - secretStoreRef: - name: vault-backend -namespace: cert-manager-dataengg-prd \ No newline at end of file diff --git a/helm-overrides/k8s-dataengg-prd-ase1/contour-cert-checker/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/contour-cert-checker/custom-values.yaml deleted file mode 100644 index b9f8329..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=k8s-dataengg-prd-ase1"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - dedicated: dataengg-devops - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dataengg-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/k8s-dataengg-prd-ase1/contour-external/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/contour-external/custom-values.yaml deleted file mode 100644 index 46139c7..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/contour-external/custom-values.yaml +++ /dev/null @@ -1,139 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 2048Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external - nodeSelector: - dedicated: contour-external - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external - nodeSelector: - dedicated: contour-external - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "50" - keda: - enabled: true - scaledown: - policies: - - periodseconds: 180 - type: Pods - value: 2 - selectpolicy: Min - stabilizationWindowSeconds: 1200 - scaleup: - policies: - - periodseconds: 15 - type: Percent - value: 20 - selectpolicy: Max - stabilizationWindowSeconds: 120 - # triggers: - # - metadata: - # desiredReplicas: "8" - # end: 50 8 * * * - # start: 15 7 * * * - # timezone: Asia/Kolkata - # type: cron - # - metadata: - # desiredReplicas: "12" - # end: 45 10 * * * - # start: 10 9 * * * - # timezone: Asia/Kolkata - # type: cron - # - metadata: - # desiredReplicas: "15" - # end: 45 10 * * * - # start: 10 10 * * * - # timezone: Asia/Kolkata - # type: cron - # - metadata: - # desiredReplicas: "18" - # end: 10 12 * * * - # start: 0 11 * * * - # timezone: Asia/Kolkata - # type: cron - # - metadata: - # desiredReplicas: "18" - # end: 55 15 * * * - # start: 45 14 * * * - # timezone: Asia/Kolkata - # type: cron - # - metadata: - # desiredReplicas: "20" - # end: 10 17 * * * - # start: 30 16 * * * - # timezone: Asia/Kolkata - # type: cron - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 8Gi - service: - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-dataengg-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-dataengg-prd-ase1/contour-internal-0/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/contour-internal-0/custom-values.yaml deleted file mode 100644 index bcfaf18..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,89 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "40" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-dataengg-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-dataengg-prd-ase1/contour-internal-1/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/contour-internal-1/custom-values.yaml deleted file mode 100644 index 083b308..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,94 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 10Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-shared - nodeSelector: - dedicated: contour-shared - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 28 - memory: 16Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int1-dataengg-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-dataengg-prd-ase1/contour-internal-intra-0/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index 9058f7f..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,87 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "40" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-dataengg-prd-ase1/contour-internal-intra-1/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index a8f3a7e..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,92 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 100% - maxUnavailable: 0 - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 10Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-shared - nodeSelector: - dedicated: contour-shared - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-intra-1 - nodeSelector: - dedicated: contour-intra-1 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 300 - targetCPU: "40" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 13 - memory: 8Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-dataengg-prd-ase1/coredns/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/coredns/custom-values.yaml deleted file mode 100644 index 4b385a8..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/coredns/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -replicaCount: 30 -communicationType: "intra" - -labels: - bu: dataengg - team: dataengg-devops - env: prd - -clusterIP: 10.137.12.2 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "dataengg-devops" - effect: "NoSchedule" - -nodeSelector: - dedicated: dataengg-devops - kubernetes.io/os: linux diff --git a/helm-overrides/k8s-dataengg-prd-ase1/coroot-node-agent/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/coroot-node-agent/custom-values.yaml deleted file mode 100644 index c0d8b53..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/coroot-node-agent/custom-values.yaml +++ /dev/null @@ -1,53 +0,0 @@ - -fullnameOverride: "coroot-node-agent-dataengg-prd" -priorityClassName: "low-priority" - -scrape: "false" # disable scraping for now - -resources: - requests: - cpu: "300m" - memory: "1Gi" - limits: - cpu: "500m" - memory: "1.5Gi" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-dataengg-contour-ext - operator: NotIn - values: - - dedicated - - key: p-dataengg-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-dataengg-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-dataengg-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-dataengg-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external \ No newline at end of file diff --git a/helm-overrides/k8s-dataengg-prd-ase1/external-secrets/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/external-secrets/custom-values.yaml deleted file mode 100644 index 6bc07fe..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/external-secrets/custom-values.yaml +++ /dev/null @@ -1,61 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Annotations to add to Pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dataengg-devops - nodeSelector: - dedicated: dataengg-devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dataengg-devops - nodeSelector: - dedicated: dataengg-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dataengg-devops - nodeSelector: - dedicated: dataengg-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" diff --git a/helm-overrides/k8s-dataengg-prd-ase1/flagger/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/flagger/custom-values.yaml deleted file mode 100644 index 0033711..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/flagger/custom-values.yaml +++ /dev/null @@ -1,63 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - dedicated: dataengg-devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dataengg-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: dataengg-devops - bu: infra - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-dataengg-rollout-service.prd-dataengg-rollout-service.svc.cluster.local" - -rolloutService: - enabled: true - apps: - prd-insights-funnel-service: yes diff --git a/helm-overrides/k8s-dataengg-prd-ase1/fluentd/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/fluentd/custom-values.yaml deleted file mode 100644 index 64eb51b..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/fluentd/custom-values.yaml +++ /dev/null @@ -1,730 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-deng-desre-fluentd-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: dataengg - team: sre - type: fluentd - service: fluentd-dataengg-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: dataengg - team: sre - type: fluentd - service: fluentd-dataengg-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: - - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - from_encoding "utf-8" - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-dataengg-prd-ase1/grafana/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/grafana/custom-values.yaml deleted file mode 100644 index c013a92..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/grafana/custom-values.yaml +++ /dev/null @@ -1,102 +0,0 @@ -# grafana: -replicas: 1 -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 10 - targetCPU: 60 - -resources: - requests: - memory: 1Gi - cpu: 1 - limits: - memory: 4Gi - cpu: 4 - -plugins: - - grafana-piechart-panel - -externalSecrets: - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/prd/infr/dvops/grafana-deng-dping-prd" - - podDisruptionBudget: - minAvailable: 1 - -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - nginx.ingress.kubernetes.io/client_header_buffer_size: "512k" - nginx.ingress.kubernetes.io/large_client_header_buffers: "4 512k" - hosts: - - 'grafana-dataengg.meesho.dev' -nodeSelector: - dedicated: "vmselect-mds" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - -envFromSecrets: - - name: grafana-dataengg-prd-secret -labels: - bu: dataengg - team: dpcon - type: grafana - service: grafana-dataengg-prd - priority: p0 - env: prd - -podLabels: - bu: dataengg - team: dpcon - type: grafana - service: p-dataplatform-grafana - priority: p0 - env: prd - -headlessService: true - -grafana.ini: - paths: - data: /var/lib/grafana/ - logs: /var/log/grafana - plugins: /var/lib/grafana/plugins - provisioning: /etc/grafana/provisioning - analytics: - check_for_updates: true - log: - mode: console - grafana_net: - url: https://grafana.net - database: - max_idle_conn: 5 - dataproxy: - timeout: 300 - security: - allow_embedding: true - server: - domain: "{{ if (and .Values.ingress.enabled .Values.ingress.hosts) }}{{ .Values.ingress.hosts | first }}{{ else }}''{{ end }}" - http_addr: "0.0.0.0" - root_url: "https://{{ if (and .Values.ingress.enabled .Values.ingress.hosts) }}{{ .Values.ingress.hosts | first }}{{ else }}''{{ end }}" - enable_gzip: true - users: - auto_assign_org_role: "Editor" - auth.proxy: - enabled: true - header_name: "X-WEBAUTH-USER" - header_property: "username" - unified_alerting: - enabled: true - # alerting: - # enabled: true diff --git a/helm-overrides/k8s-dataengg-prd-ase1/ingress-nginx-external/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/ingress-nginx-external/custom-values.yaml deleted file mode 100644 index deb02e3..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/ingress-nginx-external/custom-values.yaml +++ /dev/null @@ -1,34 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - resources: - requests: - cpu: 100m - memory: 768Mi - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 120 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 70 - ingressClass: nginx-external - ingressClassByName: true - watchIngressWithoutClass: false - ingressClassResource: - controllerValue: k8s.io/ingress-nginx-external - name: nginx-external - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dataengg-external-prd"}}}' - nodeSelector: - dedicated: nginx-external - tolerations: - - key: "dedicated" - operator: "Equal" - value: "nginx-external" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-dataengg-prd-ase1/ingress-nginx-internal/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/ingress-nginx-internal/custom-values.yaml deleted file mode 100644 index d142c2b..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/ingress-nginx-internal/custom-values.yaml +++ /dev/null @@ -1,33 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - resources: - requests: - cpu: 100m - memory: 768Mi - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 120 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 70 - extraArgs: - enable-ssl-passthrough: true - tcp-services-configmap: nginx-dataengg-internal/tcp-services - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dataengg-internal-prd"}}}' - nodeSelector: - dedicated: nginx-internal - tolerations: - - key: "dedicated" - operator: "Equal" - value: "nginx-internal" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-dataengg-prd-ase1/ingress-nginx-secured/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/ingress-nginx-secured/custom-values.yaml deleted file mode 100644 index bda70dc..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/ingress-nginx-secured/custom-values.yaml +++ /dev/null @@ -1,33 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - resources: - requests: - cpu: 100m - memory: 768Mi - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 120 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 70 - extraArgs: - enable-ssl-passthrough: true - tcp-services-configmap: nginx-dataengg-secured/tcp-services - ingressClassResource: - name: nginx-secured - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dataengg-secured-prd"}}}' - nodeSelector: - dedicated: nginx-secured - tolerations: - - key: "dedicated" - operator: "Equal" - value: "nginx-secured" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-dataengg-prd-ase1/keda/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/keda/custom-values.yaml deleted file mode 100644 index ad0e7f5..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/keda/custom-values.yaml +++ /dev/null @@ -1,61 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: dataengg-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "dataengg-devops" - effect: "NoSchedule" - podLabels: - bu: "dataengg" - team: "dataengg-devops" - metricsAdapter: - bu: "dataengg" - team: "dataengg-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 2000Mi - requests: - cpu: 400m - memory: 300Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 30m - memory: 100Mi - metricServer: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 200m - memory: 200Mi - podAnnotations: - # -- Pod annotations for KEDA operator - keda: - prometheus.io/path: /metrics - prometheus.io/port: '8080' - prometheus.io/scrape: 'true' - # -- Pod annotations for KEDA Metrics Adapter - metricsAdapter: - prometheus.io/path: /metrics - prometheus.io/port: '8080' - prometheus.io/scrape: 'true' - # -- Pod annotations for KEDA Admission webhooks - webhooks: - prometheus.io/path: /metrics - prometheus.io/port: '8080' - prometheus.io/scrape: 'true' - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' diff --git a/helm-overrides/k8s-dataengg-prd-ase1/kube-dns/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/kube-dns/custom-values.yaml deleted file mode 100644 index d5b970f..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.137.12.2"],"prd.mrouter.int.svc.cluster.local":["10.137.12.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.137.12.2"]} diff --git a/helm-overrides/k8s-dataengg-prd-ase1/kube-events/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/kube-events/custom-values.yaml deleted file mode 100644 index 3a16e5a..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-datengg-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - dedicated: dataengg-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: dataengg-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: dataengg-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: dataengg-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: dataengg-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: dataengg-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-dataengg-prd-ase1/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/kube-state-metrics/custom-values.yaml deleted file mode 100644 index cdee8a8..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-dataengg-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-dataengg-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "kube-state-metrics-dataengg-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "vmselect-mds" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 60m - memory: 800Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-dataengg-prd-ase1/kubectl-mcp-server/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/kubectl-mcp-server/custom-values.yaml deleted file mode 100644 index bf61309..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/kubectl-mcp-server/custom-values.yaml +++ /dev/null @@ -1,117 +0,0 @@ -fullnameOverride: "kubectl-mcp-server" - -replicas: 1 - -image: - # In-house hardened build (helm-templates/kubectl-mcp-server/docker/) — - # NOT the upstream Docker Hub image. 266→18 HIGH/CRIT vulns, non-root, - # multi-stage slim base, kubectl v1.33.12 / helm v3.21.0. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/prd/devop/kubectl-mcp-server - tag: "v2" - pullPolicy: IfNotPresent - -labels: - bu: infra - team: devops - service: kubectl-mcp-server - env: prd - -serviceAccount: - create: true - annotations: {} - -rbac: - create: true - -podAnnotations: {} - -podSecurityContext: {} - -securityContext: - runAsNonRoot: true - runAsUser: 1000 - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -priorityClassName: "" - -nodeSelector: - dedicated: "dataengg-devops" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "dataengg-devops" - effect: NoSchedule - -affinity: {} - -resources: - requests: - cpu: 250m - memory: 512Mi - limits: - cpu: 1000m - memory: 1Gi - -mcp: - mode: single - transport: http - host: "0.0.0.0" - port: 8000 - -auth: - allowAnonymous: false - -externalSecrets: - enabled: true - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/prd/cntr/devop/kubectl-mcp-server-dataengg" - -# tcpSocket on purpose — streamable-http transport exposes only /mcp, -# there is no /health route (see chart values.yaml note). -livenessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - -readinessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - -service: - port: 8000 - type: ClusterIP - -# Contour standard (2.0.0-style): renders HTTPProxy parent/child + -intra -# instead of a networking.k8s.io Ingress. -createContourGateway: true - -ingress: - enabled: true - ingressClassName: contour-internal-1 - servicePortNumber: 8000 - hosts: - - host: kubectl-mcp-server-dataengg.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 diff --git a/helm-overrides/k8s-dataengg-prd-ase1/kubernetes-dashboard/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/kubernetes-dashboard/custom-values.yaml deleted file mode 100644 index 670d841..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/kubernetes-dashboard/custom-values.yaml +++ /dev/null @@ -1,442 +0,0 @@ -# Copyright 2017 The Kubernetes Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -# General configuration shared across resources -app: - # Mode determines if chart should deploy a full Dashboard with all containers or just the API. - # - dashboard - deploys all the containers - # - api - deploys just the API - mode: 'dashboard' - image: - pullPolicy: IfNotPresent - pullSecrets: [] - scheduling: - # Node labels for pod assignment - # Ref: https://kubernetes.io/docs/user-guide/node-selection/ - nodeSelector: {} - security: - # Allow overriding csrfKey used by API/Auth containers. - # It has to be base64 encoded random 256 bytes string. - # If empty, it will be autogenerated. - csrfKey: ~ - # SecurityContext to be added to pods - # To disable set the following configuration to null: - # securityContext: null - securityContext: - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - # ContainerSecurityContext to be added to containers - # To disable set the following configuration to null: - # containerSecurityContext: null - containerSecurityContext: - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - runAsUser: 1001 - runAsGroup: 2001 - capabilities: - drop: ["ALL"] - # Pod Disruption Budget configuration - # Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - minAvailable: 0 - maxUnavailable: 0 - networkPolicy: - enabled: false - ingressDenyAll: false - # Raw network policy spec that overrides predefined spec - # Example: - # spec: - # egress: - # - ports: - # - port: 123 - spec: {} - - # Common labels & annotations shared across all deployed resources - labels: {} - annotations: {} - # Common priority class used for all deployed resources - priorityClassName: null - settings: - ## Global dashboard settings - global: - # # Cluster name that appears in the browser window title if it is set - # clusterName: "" - # # Max number of items that can be displayed on each list page - # itemsPerPage: 10 - # # Max number of labels that are displayed by default on most views. - # labelsLimit: 3 - # # Number of seconds between every auto-refresh of logs - # logsAutoRefreshTimeInterval: 5 - # # Number of seconds between every auto-refresh of every resource. Set 0 to disable - # resourceAutoRefreshTimeInterval: 10 - # # Hide all access denied warnings in the notification panel - # disableAccessDeniedNotifications: false - # # Hide all namespaces option in namespace selection dropdown to avoid accidental selection in large clusters thus preventing OOM errors - # hideAllNamespaces: false - # # Namespace that should be selected by default after logging in. - # defaultNamespace: default - # # List of namespaces that should be presented to user without namespace list privileges. - # namespaceFallbackList: - # - default - ## Pinned resources that will be displayed in dashboard's menu - pinnedResources: [] - # - kind: customresourcedefinition - # # Fully qualified name of a CRD - # name: prometheus.monitoring.coreos.com - # # Display name - # displayName: Prometheus - # # Is this CRD namespaced? - # namespaced: true - ingress: - enabled: false - hosts: - # Keep 'localhost' host only if you want to access Dashboard using 'kubectl port-forward ...' on: - # https://localhost:8443 - - localhost - # - kubernetes.dashboard.domain.com - ingressClassName: internal-nginx - # Use only if your ingress controllers support default ingress classes. - # If set to true ingressClassName will be ignored and not added to the Ingress resources. - # It should fall back to using IngressClass marked as the default. - useDefaultIngressClass: false - # This will append our Ingress with annotations required by our default configuration. - # nginx.ingress.kubernetes.io/backend-protocol: "HTTPS" - # nginx.ingress.kubernetes.io/ssl-passthrough: "true" - # nginx.ingress.kubernetes.io/ssl-redirect: "true" - useDefaultAnnotations: true - pathType: ImplementationSpecific - # If path is not the default (/), rewrite-target annotation will be added to the Ingress. - # It allows serving Kubernetes Dashboard on a sub-path. Make sure that the configured path - # does not conflict with gateway route configuration. - path: / - issuer: - name: selfsigned - # Scope determines what kind of issuer annotation will be used on ingress resource - # - default - adds 'cert-manager.io/issuer' - # - cluster - adds 'cert-manager.io/cluster-issuer' - # - disabled - disables cert-manager annotations - scope: default - tls: - enabled: true - # If provided it will override autogenerated secret name - secretName: "" - labels: {} - annotations: {} - # Use the following toleration if Dashboard can be deployed on a tainted control-plane nodes - # - key: node-role.kubernetes.io/control-plane - # effect: NoSchedule - tolerations: [] - affinity: {} - -auth: - role: auth - image: - repository: docker.io/kubernetesui/dashboard-auth - tag: 1.2.2 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: auth - containerPort: 8000 - protocol: TCP - args: [] - env: [] - volumeMounts: - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for Auth related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -# API deployment configuration -api: - role: api - image: - repository: docker.io/kubernetesui/dashboard-api - tag: 1.10.1 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: api - containerPort: 8000 - protocol: TCP - # Additional container arguments - # Full list of arguments: https://github.com/kubernetes/dashboard/blob/master/docs/common/arguments.md - # args: - # - --system-banner="Welcome to the Kubernetes Dashboard" - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store exec logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for API related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -# WEB UI deployment configuration -web: - role: web - image: - repository: docker.io/kubernetesui/dashboard-web - tag: 1.6.0 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: web - containerPort: 8000 - protocol: TCP - # Additional container arguments - # Full list of arguments: https://github.com/kubernetes/dashboard/blob/master/docs/common/arguments.md - # args: - # - --system-banner="Welcome to the Kubernetes Dashboard" - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for WEB UI related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -### Metrics Scraper -### Container to scrape, store, and retrieve a window of time from the Metrics Server. -### refs: https://github.com/kubernetes/dashboard/tree/master/modules/metrics-scraper -metricsScraper: - enabled: true - role: metrics-scraper - image: - repository: docker.io/kubernetesui/dashboard-metrics-scraper - tag: 1.2.1 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - containerPort: 8000 - protocol: TCP - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - livenessProbe: - httpGet: - scheme: HTTP - path: / - port: 8000 - initialDelaySeconds: 30 - timeoutSeconds: 30 - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for Metrics Scraper related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -## Optional Metrics Server sub-chart configuration -## Enable this if you don't already have metrics-server enabled on your cluster and -## want to use it with dashboard metrics-scraper -## refs: -## - https://github.com/kubernetes-sigs/metrics-server -## - https://github.com/kubernetes-sigs/metrics-server/tree/master/charts/metrics-server -metrics-server: - enabled: false - args: - - --kubelet-preferred-address-types=InternalIP - - --kubelet-insecure-tls - -## Required Kong sub-chart with DBless configuration to act as a gateway -## for our all containers. -kong: - enabled: true - admin: - tls: - enabled: false - ## Configuration reference: https://docs.konghq.com/gateway/3.6.x/reference/configuration - env: - dns_order: LAST,A,CNAME,AAAA,SRV - plugins: 'off' - nginx_worker_processes: 1 - ingressController: - enabled: false - manager: - enabled: false - dblessConfig: - configMap: kong-dbless-config - proxy: - type: ClusterIP - http: - enabled: true - -## Optional Cert Manager sub-chart configuration -## Enable this if you don't already have cert-manager enabled on your cluster. -cert-manager: - enabled: false - installCRDs: true - -## Optional Nginx Ingress sub-chart configuration -## Enable this if you don't already have nginx-ingress enabled on your cluster. -nginx: - enabled: false - controller: - electionID: ingress-controller-leader - ingressClassResource: - name: internal-nginx - default: false - controllerValue: k8s.io/internal-ingress-nginx - service: - type: ClusterIP - -## Extra configurations: -## - manifests -## - predefined roles -## - prometheus -## - etc... -extras: - # Extra Kubernetes manifests to be deployed - # manifests: - # - apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: additional-configmap - # data: - # mykey: myvalue - manifests: [] - serviceMonitor: - # Whether to create a Prometheus Operator service monitor. - enabled: false - # Here labels can be added to the serviceMonitor - labels: {} - # Here annotations can be added to the serviceMonitor - annotations: {} - # metrics.serviceMonitor.metricRelabelings Specify Metric Relabelings to add to the scrape endpoint - # ref: https://github.com/coreos/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # metrics.serviceMonitor.relabelings [array] Prometheus relabeling rules - relabelings: [] - # ServiceMonitor connection scheme. Defaults to HTTPS. - scheme: https - # ServiceMonitor connection tlsConfig. Defaults to {insecureSkipVerify:true}. - tlsConfig: - insecureSkipVerify: true diff --git a/helm-overrides/k8s-dataengg-prd-ase1/kyverno/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/kyverno/custom-values.yaml deleted file mode 100644 index e1a08cd..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/kyverno/custom-values.yaml +++ /dev/null @@ -1,2240 +0,0 @@ -global: - - # -- Internal settings used with `helm template` to generate install manifest - # @ignored - templating: - enabled: false - debug: false - version: ~ - - image: - # -- (string) Global value that allows to set a single image registry across all deployments. - # When set, it will override any values set under `.image.registry` across the chart. - registry: ~ - # -- (list) Global list of Image pull secrets - # When set, it will override any values set under `imagePullSecrets` under different components across the chart. - imagePullSecrets: [] - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - caCertificates: - # -- Global CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - # Individual controller values will override this global value - data: ~ - - # -- Global value to set single volume to be mounted for CA certificates for all deployments. - # Not used when `.Values.global.caCertificates.data` is defined - # Individual controller values will override this global value - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Additional container environment variables to apply to all containers and init containers - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Global node labels for pod assignment. Non-global values will override the global value. - nodeSelector: - dedicated: dataengg-kyverno - - # -- Global List of node taints to tolerate. Non-global values will override the global value. - tolerations: - - key: "dedicated" - operator: "Equal" - value: "dataengg-kyverno" - effect: "NoSchedule" - -# -- (string) Override the name of the chart -nameOverride: ~ - -# -- (string) Override the expanded name of the chart -fullnameOverride: ~ - -# -- (string) Override the namespace the chart deploys to -namespaceOverride: ~ - -upgrade: - # -- Upgrading from v2 to v3 is not allowed by default, set this to true once changes have been reviewed. - fromV2: false - -apiVersionOverride: - # -- (string) Override api version used to create `PodDisruptionBudget`` resources. - # When not specified the chart will check if `policy/v1/PodDisruptionBudget` is available to - # determine the api version automatically. - podDisruptionBudget: ~ - -rbac: - roles: - # -- Aggregate ClusterRoles to Kubernetes default user-facing roles. For more information, see [User-facing roles](https://kubernetes.io/docs/reference/access-authn-authz/rbac/#user-facing-roles) - aggregate: - admin: true - view: true - -# Use openreports.io as the API group for reporting -openreports: - # -- Enable OpenReports feature in controllers - enabled: false - # -- Whether to install CRDs from the upstream OpenReports chart. Setting this to true requires enabled to also be true. - installCrds: false - -# CRDs configuration -crds: - - # -- Whether to have Helm install the Kyverno CRDs, if the CRDs are not installed by Helm, they must be added before policies can be created - install: true - - reportsServer: - # -- Kyverno reports-server is used in your cluster - enabled: false - - groups: - - # -- Install CRDs in group `kyverno.io` - kyverno: - cleanuppolicies: true - clustercleanuppolicies: true - clusterpolicies: true - globalcontextentries: true - policies: true - policyexceptions: true - updaterequests: true - - # -- Install CRDs in group `policies.kyverno.io` - policies: - validatingpolicies: true - policyexceptions: true - imagevalidatingpolicies: true - namespacedimagevalidatingpolicies: true - mutatingpolicies: true - generatingpolicies: true - deletingpolicies: true - namespaceddeletingpolicies: true - namespacedvalidatingpolicies: true - - # -- Install CRDs in group `reports.kyverno.io` - reports: - clusterephemeralreports: true - ephemeralreports: true - - # -- Install CRDs in group `wgpolicyk8s.io` - wgpolicyk8s: - clusterpolicyreports: false ##lk - policyreports: false ##lk - - # -- Additional CRDs annotations (Replace=true avoids 262144-byte annotation limit on large CRDs) - annotations: - argocd.argoproj.io/sync-options: Replace=true - # strategy.spinnaker.io/replace: 'true' - - # -- Additional CRDs labels - customLabels: {} - - migration: - - # -- Enable CRDs migration using helm post upgrade hook - enabled: true - - # -- Resources to migrate - resources: - - cleanuppolicies.kyverno.io - - clustercleanuppolicies.kyverno.io - - clusterpolicies.kyverno.io - - globalcontextentries.kyverno.io - - policies.kyverno.io - - policyexceptions.kyverno.io - - updaterequests.kyverno.io - - deletingpolicies.policies.kyverno.io - - generatingpolicies.policies.kyverno.io - - imagevalidatingpolicies.policies.kyverno.io - - namespacedimagevalidatingpolicies.policies.kyverno.io - - mutatingpolicies.policies.kyverno.io - - namespaceddeletingpolicies.policies.kyverno.io - - namespacedvalidatingpolicies.policies.kyverno.io - - policyexceptions.policies.kyverno.io - - validatingpolicies.policies.kyverno.io - - image: - # -- (string) Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- (string) Image repository - repository: kyverno/kyverno-cli - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- (string) Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podResources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -# Configuration -config: - - # -- Create the configmap. - create: true - - # -- Preserve the configmap settings during upgrade. - preserve: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - # -- Enable registry mutation for container images. Enabled by default. - enableDefaultRegistryMutation: true - - # -- The registry hostname used for the image mutation. - defaultRegistry: docker.io - - # -- Exclude groups - excludeGroups: - - system:nodes - - # -- Exclude usernames - excludeUsernames: [] - # - '!system:kube-scheduler' - - # -- Exclude roles - excludeRoles: [] - - # -- Exclude roles - excludeClusterRoles: [] - - # -- Generate success events. - generateSuccessEvents: false - - # -- Resource types to be skipped by the Kyverno policy engine. - # Make sure to surround each entry in quotes so that it doesn't get parsed as a nested YAML list. - # These are joined together without spaces, run through `tpl`, and the result is set in the config map. - # @default -- See [values.yaml](values.yaml) - resourceFilters: - - '[Event,*,*]' - # - '[*/*,kube-system,*]' - - '[*/*,kube-public,*]' - - '[*/*,kube-node-lease,*]' - - '[Node,*,*]' - - '[Node/?*,*,*]' - - '[APIService,*,*]' - - '[APIService/?*,*,*]' - - '[TokenReview,*,*]' - - '[SubjectAccessReview,*,*]' - - '[SelfSubjectAccessReview,*,*]' - - '[Binding,*,*]' - - '[Pod/binding,*,*]' - - '[ReplicaSet,*,*]' - - '[ReplicaSet/?*,*,*]' - - '[EphemeralReport,*,*]' - - '[ClusterEphemeralReport,*,*]' - # exclude resources from the chart - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:additional]' - - '[ClusterRoleBinding,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.configMapName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.metricsConfigMapName" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Job,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[Job/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.admission-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.background-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - # -- Sets the threshold for the total number of UpdateRequests generated for mutateExisitng and generate policies. - updateRequestThreshold: 1000 - - # -- Defines the `namespaceSelector`/`objectSelector` in the webhook configurations. - # The Kyverno namespace is excluded if `excludeKyvernoNamespace` is `true` (default) - webhooks: - # Exclude namespaces - # namespaceSelector: lk-s2 - # matchExpressions: - # - key: kubernetes.io/metadata.name - # operator: NotIn - # values: - # - kube-systemm lk-s2 - # Exclude objects - # objectSelector: - # matchExpressions: - # - key: webhooks.kyverno.io/exclude - # operator: DoesNotExist - - # -- Defines annotations to set on webhook configurations. - webhookAnnotations: - # Example to disable admission enforcer on AKS: - # 'admissions.enforcer/disabled': 'true' lk-s3 - - # -- Defines labels to set on webhook configurations. - webhookLabels: {} - # Example to adopt webhook resources in ArgoCD: - # 'argocd.argoproj.io/instance': 'kyverno' - - # -- Defines match conditions to set on webhook configurations (requires Kubernetes 1.27+). - matchConditions: [] - - # -- Exclude Kyverno namespace - # Determines if default Kyverno namespace exclusion is enabled for webhooks and resourceFilters - excludeKyvernoNamespace: true - - # -- resourceFilter namespace exclude - # Namespaces to exclude from the default resourceFilters - resourceFiltersExcludeNamespaces: [] - - # -- resourceFilters exclude list - # Items to exclude from config.resourceFilters - resourceFiltersExclude: [] - - # -- resourceFilter namespace include - # Namespaces to include to the default resourceFilters - resourceFiltersIncludeNamespaces: [] - - # -- resourceFilters include list - # Items to include to config.resourceFilters - resourceFiltersInclude: [] - -# Metrics configuration -metricsConfig: - - # -- Create the configmap. - create: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - namespaces: - - # -- List of namespaces to capture metrics for. - include: [] - - # -- list of namespaces to NOT capture metrics for. - exclude: [] - - # -- (string) Rate at which metrics should reset so as to clean up the memory footprint of kyverno metrics, if you might be expecting high memory footprint of Kyverno's metrics. Default: 0, no refresh of metrics. WARNING: This flag is not working since Kyverno 1.8.0 - metricsRefreshInterval: ~ - # metricsRefreshInterval: 24h - - # -- (list) Configures the bucket boundaries for all Histogram metrics, changing this configuration requires restart of the kyverno admission controller - bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 15, 20, 25, 30] - - # -- (map) Configures the exposure of individual metrics, by default all metrics and all labels are exported, changing this configuration requires restart of the kyverno admission controller - metricsExposure: - kyverno_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_image_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_mutating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_generating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_admission_review_duration_seconds: - # enabled: false - disabledLabelDimensions: ["resource_namespace"] - kyverno_policy_rule_info_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_policy_results_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_admission_requests_total: - disabledLabelDimensions: ["resource_namespace"] - kyverno_cleanup_controller_deletedobjects_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - -# -- Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -imagePullSecrets: {} - # regcred: - # registry: foo.example.com - # username: foobar - # password: secret - # regcred2: - # registry: bar.example.com - # username: barbaz - # password: secret2 - -# -- Existing Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -existingImagePullSecrets: [] - # - test-registry - # - other-test-registry - -# Tests configuration -test: - # -- Sleep time before running test - sleep: 20 - - image: - # -- (string) Image registry - registry: curlimages - # -- Image repository - repository: curl - # -- Image tag - # Defaults to `latest` if omitted - tag: '8.10.1' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - # -- Security context for the test containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- Additional Pod annotations - podAnnotations: {} - - # -- List of node taints to tolerate - tolerations: [] - -# -- Additional labels -customLabels: {} - - -webhooksCleanup: - # -- Create a helm pre-delete hook to cleanup webhooks. - enabled: true - - autoDeleteWebhooks: - # -- Allow webhooks controller to delete webhooks using finalizers - enabled: false - - image: - # -- (string) Image registry - registry: registry.k8s.io - # -- Image repository - repository: kubectl - # -- Image tag - # Defaults to `latest` if omitted - tag: 'v1.32.7' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -grafana: - # -- Enable grafana dashboard creation. - enabled: false - - # -- Configmap name template. - configMapName: '{{ include "kyverno.fullname" . }}-grafana' - - # -- (string) Namespace to create the grafana dashboard configmap. - # If not set, it will be created in the same namespace where the chart is deployed. - namespace: ~ - - # -- Grafana dashboard configmap annotations. - annotations: {} - - # -- Grafana dashboard configmap labels - labels: - grafana_dashboard: "1" - - # -- create GrafanaDashboard custom resource referencing to the configMap. - # according to https://grafana-operator.github.io/grafana-operator/docs/examples/dashboard_from_configmap/readme/ - grafanaDashboard: - create: false - folder: kyverno - allowCrossNamespaceImport: true - matchLabels: - dashboards: "grafana" - -# Features configuration -features: - admissionReports: - # -- Enables the feature - enabled: true - aggregateReports: - # -- Enables the feature - enabled: true - policyReports: - # -- Enables the feature - enabled: false ##lk - validatingAdmissionPolicyReports: - # -- Enables the feature - enabled: true - mutatingAdmissionPolicyReports: - # -- Enables the feature - enabled: false - reporting: - # -- Enables the feature - validate: true - # -- Enables the feature - mutate: true - # -- Enables the feature - mutateExisting: true - # -- Enables the feature - imageVerify: true - # -- Enables the feature - generate: true - autoUpdateWebhooks: - # -- Enables the feature - enabled: true - backgroundScan: - # -- Enables the feature - enabled: true - # -- Number of background scan workers - backgroundScanWorkers: 2 - # -- Background scan interval - backgroundScanInterval: 1h - # -- Skips resource filters in background scan - skipResourceFilters: true - configMapCaching: - # -- Enables the feature - enabled: true - controllerRuntimeMetrics: - # -- Bind address for controller-runtime metrics (use "0" to disable it) - bindAddress: ":8080" - deferredLoading: - # -- Enables the feature - enabled: true - dumpPayload: - # -- Enables the feature - enabled: false - forceFailurePolicyIgnore: - # -- Enables the feature - enabled: false - generateValidatingAdmissionPolicy: - # -- Enables the feature - enabled: true - generateMutatingAdmissionPolicy: - # -- Enables the feature - enabled: false - dumpPatches: - # -- Enables the feature - enabled: false - globalContext: - # -- Maximum allowed response size from API Calls. A value of 0 bypasses checks (not recommended) - maxApiCallResponseLength: 2000000 - logging: - # -- Logging format - format: text - # -- Logging verbosity - verbosity: 2 - omitEvents: - # -- Events which should not be emitted (possible values `PolicyViolation`, `PolicyApplied`, `PolicyError`, and `PolicySkipped`) - eventTypes: - - PolicyApplied - - PolicySkipped - # - PolicyViolation - # - PolicyError - policyExceptions: - # -- Enables the feature - enabled: true - # -- Restrict policy exceptions to a single namespace - # Set to "*" to allow exceptions in all namespaces - namespace: '' - protectManagedResources: - # -- Enables the feature - enabled: false - registryClient: - # -- Allow insecure registry - allowInsecure: false - # -- Enable registry client helpers - credentialHelpers: - - default - - google - - amazon - - azure - - github - ttlController: - # -- Reconciliation interval for the label based cleanup manager - reconciliationInterval: 1m - tuf: - # -- Enables the feature - enabled: false - # -- (string) Path to Tuf root - root: ~ - # -- (string) Raw Tuf root - rootRaw: ~ - # -- (string) Tuf mirror - mirror: ~ - -# Admission controller configuration -admissionController: - autoscaling: - # -- Enable horizontal pod autoscaling - enabled: true - - # -- Minimum number of pods - minReplicas: 1 - - # -- Maximum number of pods - maxReplicas: 10 - - # -- Target CPU utilization percentage - targetCPUUtilizationPercentage: 80 - - # -- Configurable scaling behavior - behavior: {} - - # -- Overrides features defined at the root level - featuresOverride: - admissionReports: - # -- Max number of admission reports allowed in flight until the admission controller stops creating new ones - backPressureThreshold: 1000 - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- The ServiceAccount name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno admission controller activities. - # This will help ensure Kyverno stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- admissionController webhook server port - # in case you are using hostNetwork: true, you might want to change the port the webhookServer is listening to - webhookServer: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - admission-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.admissionController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - initContainer: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyvernopre - # -- (string) Image tag - # If missing, defaults to image.tag - tag: ~ - # -- (string) Image pull policy - # If missing, defaults to image.pullPolicy - pullPolicy: ~ - - resources: - # -- Pod resource limits - # limits: - # cpu: 100m - # memory: 256Mi - # -- Pod resource requests - requests: - cpu: 1000m - memory: 1Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - container: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyverno - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - resources: - # -- Pod resource limits - # limits: - # memory: 384Mi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Array of extra init containers - extraInitContainers: [] - # - name: init-container - # image: busybox - # command: ['sh', '-c', 'echo Hello'] - - # -- Array of extra containers to run alongside kyverno - extraContainers: [] - # - name: myapp-container - # image: busybox - # command: ['sh', '-c', 'echo Hello && sleep 3600'] - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Kyverno's metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Background controller configuration -backgroundController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable background controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: - - apiGroups: - - networking.k8s.io - resources: - - ingresses - - ingressclasses - - networkpolicies - verbs: - - create - - update - - patch - - delete - - apiGroups: - - rbac.authorization.k8s.io - resources: - - rolebindings - - roles - verbs: - - create - - update - - patch - - delete - - apiGroups: - - '' - resources: - - configmaps - - resourcequotas - - limitranges - verbs: - - create - - update - - patch - - delete - - apiGroups: - - resource.k8s.io - resources: - - resourceclaims - - resourceclaimtemplates - verbs: - - create - - delete - - update - - patch - - deletecollection - - apiGroups: - - autoscaling - resources: - - horizontalpodautoscalers - verbs: - - get - - list - - watch - - update - - patch - - apiGroups: - - apps - resources: - - deployments - - statefulsets - - replicasets - verbs: - - get - - list - - watch - - update - - patch - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - # - patch - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/background-controller - # -- Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - # limits: - # memory: 128Mi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - background-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.backgroundController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - # -- backgroundController server port - # in case you are using hostNetwork: true, you might want to change the port the backgroundController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Cleanup controller configuration -cleanupController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable cleanup controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - delete - # - list - # - watch - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/cleanup-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- cleanupController server port - # in case you are using hostNetwork: true, you might want to change the port the cleanupController is listening to - server: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - # limits: - # memory: 128Mi - # -- Pod resource requests - requests: - cpu: 1100m - memory: 1.2Gi - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - cleanup-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `service.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Reports controller configuration -reportsController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable reports controller. - enabled: false ##lk - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/reports-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno reports controller activities. - # This will help ensure Kyverno reports stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - memory: 128Mi - # -- Pod resource requests - requests: - cpu: 100m - memory: 64Mi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - reports-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.reportsController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- (string) Service node port. - # Only used if `type` is `NodePort`. - nodePort: ~ - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- (string) Traces receiver address - address: ~ - # -- (string) Traces receiver port - port: ~ - # -- (string) Traces receiver credentials - creds: ~ - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- (string) Otel collector endpoint - collector: ~ - # -- (string) Otel collector credentials - creds: ~ - - # -- reportsController server port - # in case you are using hostNetwork: true, you might want to change the port the reportsController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - - # -- Enable sanity check for reports CRDs - sanityChecks: true diff --git a/helm-overrides/k8s-dataengg-prd-ase1/kyverno/policies/protect-namespace.yaml b/helm-overrides/k8s-dataengg-prd-ase1/kyverno/policies/protect-namespace.yaml deleted file mode 100644 index 32a4a88..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/kyverno/policies/protect-namespace.yaml +++ /dev/null @@ -1,40 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: protect-critical-namespaces -spec: - validationFailureAction: Enforce - background: false - rules: - - name: deny-deletion-of-critical-namespaces - match: - any: - - resources: - kinds: - - Namespace - names: - - keda-dataengg-prd - - contour-internal-0-dataengg-prd - - contour-internal-0-dataengg-prd-intra - - contour-external-dataengg-prd - - external-secrets-dataengg-prd - - flagger-dataengg-prd - - victoriametrics - - kube-system - - monitoring - - telegraf-operator - - loadtester - - opentelemetry - - fluentd - - kube-events - - observability - - contour-cert-checker-ns - - aurva-dataplane - validate: - message: "Deletion of critical namespaces is not allowed." - deny: - conditions: - any: - - key: "{{request.operation}}" - operator: Equals - value: DELETE diff --git a/helm-overrides/k8s-dataengg-prd-ase1/kyverno/policies/restrict-replicas.yaml b/helm-overrides/k8s-dataengg-prd-ase1/kyverno/policies/restrict-replicas.yaml deleted file mode 100644 index d456985..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/kyverno/policies/restrict-replicas.yaml +++ /dev/null @@ -1,29 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: restrict-minimum-replicas-selected-ns -spec: - validationFailureAction: Enforce - background: false - rules: - - name: min-replicas-deployments-statefulsets - match: - resources: - kinds: - - Deployment - - StatefulSet - - Deployment/scale - - StatefulSet/scale - names: - - coredns - namespaces: - - kube-system - operations: - - CREATE - - UPDATE - validate: - message: "Deployments and StatefulSets must have at least 5 replicas in this namespace." - anyPattern: - # Case 1: replicas explicitly set and >= 5 - - spec: - replicas: ">=5" diff --git a/helm-overrides/k8s-dataengg-prd-ase1/loadtester/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/loadtester/custom-values.yaml deleted file mode 100644 index c329edc..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/loadtester/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -replicaCount: 1 - -Namespace: loadtester - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/dev/devops/flagger-loadtester - tag: 0.30.0 - pullPolicy: IfNotPresent - pullSecret: - -podLabels: - bu: dataengg - team: dataengg-shared - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - -podPriorityClassName: "" - -logLevel: info -cmd: - timeout: 1h - namespaceRegexp: "" - -nameOverride: "loadtester" -fullnameOverride: "" - -env: [] - -service: - type: ClusterIP - port: 80 - -resources: - requests: - cpu: 10m - memory: 64Mi - -volumes: [] -volumeMounts: [] - -nodeSelector: - dedicated: dataengg-devops - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "dataengg-devops" - effect: "NoSchedule" - -affinity: {} - -ingress: - slowStart: - enabled: false - enabled: true - ingressClassName: contour-internal-1 - servicePort: http - hosts: - - host: dataengg-flagger-loadtester.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / -createContourGateway: true -contourResponseTimeout: false -rbac: - # rbac.create: `true` if rbac resources should be created - create: true - # rbac.scope: `cluster` to create cluster-scope rbac resources (ClusterRole/ClusterRoleBinding) - # otherwise, namespace-scope rbac resources will be created (Role/RoleBinding) - scope: - # rbac.rules: array of rules to apply to the role. example: - # rules: - # - apiGroups: [""] - # resources: ["pods"] - # verbs: ["list", "get"] - rules: [] - -# name of an existing service account to use - if not creating rbac resources -serviceAccountName: "" - -# App Mesh virtual node settings (to be used for AppMesh v1beta1) -meshName: "" -#backends: -# - app1.namespace -# - app2.namespace - -# App Mesh virtual node settings (to be used for AppMesh v1beta2) -appmesh: - enabled: false - backends: - - podinfo - - podinfo-canary - -#Istio virtual service and gatway settings. TLS secrets should be in namespace before enbaled it. ( secret format loadtester.fullname ) -istio: - enabled: false - host: flagger-loadtester.flagger - gateway: - enabled: false - tls: - enabled: false - httpsRedirect: false - -# when enabled, it will add a security context for the loadtester pod -securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 100 - runAsGroup: 101 - -podDisruptionBudget: - enabled: false - minAvailable: 1 diff --git a/helm-overrides/k8s-dataengg-prd-ase1/opentelemetry-coralogix/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/opentelemetry-coralogix/custom-values.yaml deleted file mode 100644 index 2c051da..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/opentelemetry-coralogix/custom-values.yaml +++ /dev/null @@ -1,284 +0,0 @@ -global: - domain: "coralogixsg.com" - defaultApplicationName: "default" - defaultSubsystemName: "nodes" - - # Old endpoint based configuration, - # please use domain instead. - traces: - endpoint: "" - metrics: - endpoint: "" - logs: - endpoint: "" - -# set distribution to openshift for openshift clusters -distribution: "" - -# Opentelemetry collector configuration -dedicatedValue: false -schedulerName: default-scheduler - -labels: - bu: "dataengg" - team: "sre" - service: "opentelemetry-dataengg-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: false - key: prd/admin/coralogix-keys - secretStoreRef: - name: vault-backend - -image: - # If you want to use the core image `otel/opentelemetry-collector`, you also need to change `command.name` value to `otelcol`. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - # repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - tag: "0.87.0" - # When digest is set to a non-empty value, images will be pulled by digest (regardless of tag value). - digest: "" -mode: daemonset -rollout: - rollingUpdate: - # When 'mode: daemonset', maxSurge cannot be used when hostPort is set for any of the ports - # maxSurge: 25% - maxUnavailable: 5 - strategy: RollingUpdate -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" -hostNetwork: true -dnsPolicy: "ClusterFirstWithHostNet" -fullnameOverride: opentelemetry-dataengg-prd - -presets: - logsCollection: - enabled: false - storeCheckpoints: false - maxRecombineLogSize: 1048576 - extraFilelogOperators: [] -# - type: recombine -# combine_field: body -# source_identifier: attributes["log.file.path"] -# is_first_entry: body matches "^(YOUR-LOGS-REGEX)" - kubernetesAttributes: - enabled: false - hostMetrics: - enabled: false - kubeletMetrics: - enabled: false - -extraEnvs: -- name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" -- name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName -config: - extensions: {} - # zpages: - # endpoint: localhost:55679 - # pprof: - # endpoint: localhost:1777 - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 4s - tls: - insecure: true - sending_queue: - enabled: true - queue_size: 6000 - num_consumers: 200 - resolver: - dns: - hostname: opentelemetry-admin-prd.opentelemetry.svc.clusterset.local - # coralogix: - # timeout: "30s" - # private_key: "${CORALOGIX_PRIVATE_KEY}" - # domain: "{{.Values.global.domain}}" - # traces: - # endpoint: "{{ .Values.global.traces.endpoint }}" - # metrics: - # endpoint: "{{ .Values.global.metrics.endpoint }}" - # logs: - # endpoint: "{{ .Values.global.logs.endpoint }}" - # application_name_attributes: - # - "k8s.namespace.name" - # - "service.namespace" - # subsystem_name_attributes: - # - "k8s.deployment.name" - # - "k8s.statefulset.name" - # - "k8s.daemonset.name" - # - "k8s.cronjob.name" - # - "k8s.job.name" - # - "k8s.container.name" - # - "k8s.node.name" - # - "service.name" - # application_name: "{{.Values.global.defaultApplicationName }}" - # subsystem_name: "{{.Values.global.defaultSubsystemName }}" - processors: - batch: - send_batch_size: 3072 - timeout: 5s - k8sattributes: - filter: - node_from_env_var: KUBE_NODE_NAME - extract: - metadata: - - "k8s.namespace.name" - - "k8s.deployment.name" - - "k8s.statefulset.name" - - "k8s.daemonset.name" - - "k8s.cronjob.name" - - "k8s.job.name" - - "k8s.pod.name" - - "k8s.node.name" - memory_limiter: null # Will get the k8s resource limits - resourcedetection/env: - detectors: ["system","env"] - timeout: 2s - override: false - # spanmetrics: - # metrics_exporter: coralogix - # dimensions: - # - name: "k8s.deployment.name" - # - name: "k8s.statefulset.name" - # - name: "k8s.daemonset.name" - # - name: "k8s.cronjob.name" - # - name: "k8s.job.name" - # - name: "k8s.container.name" - # - name: "k8s.node.name" - # - name: "k8s.namespace.name" - receivers: - otlp: - protocols: - grpc: - endpoint: ${MY_POD_IP}:4317 - http: - endpoint: ${MY_POD_IP}:4318 - zipkin: - endpoint: ${MY_POD_IP}:9411 - jaeger: - protocols: - grpc: - endpoint: ${MY_POD_IP}:14250 - thrift_http: - endpoint: ${MY_POD_IP}:14268 - thrift_compact: - endpoint: ${MY_POD_IP}:6831 - thrift_binary: - endpoint: ${MY_POD_IP}:6832 - prometheus: - config: - scrape_configs: - - job_name: opentelemetry-collector - scrape_interval: 30s - static_configs: - - targets: - - ${MY_POD_IP}:8888 - service: - extensions: - # - zpages - # - pprof - - health_check -# - memory_ballast - telemetry: - metrics: - address: ${env:MY_POD_IP}:8888 - pipelines: - traces: - exporters: - - loadbalancing - processors: - - memory_limiter - # - spanmetrics - - batch - receivers: - - otlp - metrics: null - logs: null -tolerations: - - operator: Exists - -resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 1 - memory: 2G - -ports: - jaeger-binary: - enabled: true - containerPort: 6832 - servicePort: 6832 - hostPort: 6832 - protocol: TCP - # In order to enable podMonitor, following part must be enabled in order to expose the required port: - # metrics: - # enabled: true - -# podMonitor: -# enabled: true - -# prometheusRule: -# enabled: true -# defaultRules: -# enabled: true - -# Annotations to be added to pod -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-dataengg-contour-ext - operator: NotIn - values: - - dedicated - - key: p-dataengg-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-dataengg-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-dataengg-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-dataengg-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert \ No newline at end of file diff --git a/helm-overrides/k8s-dataengg-prd-ase1/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index 428e2ed..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,134 +0,0 @@ -fullnameOverride: opentelemetry-dataengg-prd - -mode: daemonset - -# priorityClassName: "system-node-critical" -# Removing since dataengg pods are going into pending state -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-dataengg-contour-ext - operator: NotIn - values: - - dedicated - - key: p-dataengg-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-dataengg-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-dataengg-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-dataengg-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external - -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 500m - memory: 256Mi - limits: - cpu: 1 - memory: 512Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "dataengg" - team: "sre" - service: "opentelemetry-dataengg-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/k8s-dataengg-prd-ase1/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index 1dfdf24..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,496 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-admin-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "node-exporter-dataengg-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-dataengg-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index 97b6538..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-dataengg-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-dataengg-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "stackdriver-exporter-dataengg-prd" - env: "prd" - priority: "p0" - type: "exporter" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-dataengg-prd-0622" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime,compute.googleapis.com/instance' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - dedicated: vmselect-mds - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-deng-desre-stackdriver-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - } - - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/k8s-dataengg-prd-ase1/telegraf-operator/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/telegraf-operator/custom-values.yaml deleted file mode 100644 index 60d7565..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,179 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - - mq-scraper: | - [[inputs.http]] - urls = ["http://mq-server.prd.meesho.int/api/v1/consumerMap"] - method = "GET" - timeout = "10s" - name_override = "mq_consumers" - data_format = "json" - json_query = "consumerMap" - fieldpass = ["value"] # or json_string_fields = ["value"] check the difference bw fieldpass and json_string_fields - tag_keys = ["applicationName", "topicName", "groupId"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "60s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "dataengg" - team: "dataengg-sre" - service: "telegraf-operator-dataengg-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - dedicated: "vmselect-mds" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false \ No newline at end of file diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml deleted file mode 100644 index c5f9f85..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml +++ /dev/null @@ -1,296 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-dataengg-prd-dr -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-dataengg-prd-dr-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent-dr - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-deng-desre-vmagent-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - - https://vminsert-infra-prd-dr.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vmagent-dataengg-prd" - env: "prd" - priority: "p0" - type: "vmagent-dr" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dataengg-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dataengg-prd-dr.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 4 - memory: 6Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent-dr" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml deleted file mode 100644 index 877fc2c..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml +++ /dev/null @@ -1,272 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 1 - -fullnameOverride: vmagent-dataengg-prd-fb -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-dataengg-prd-fb-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - -# Remove topology spread constraints -topologySpreadConstraints: [] - -# Modify deployment strategy -deployment: - enabled: true - -# Remove HPA -horizontalPodAutoscaler: - enabled: false - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-infr-sre-vmagent-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -# Remove pod disruption budget -podDisruptionBudget: - enabled: false - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-dataengg.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vmsinglenode-infra-prd.victoriametrics.svc.clusterset.local:8428/api/v1/write - - https://vmsinglenode-infra-prd.meeshogcp.in/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dataengg-fb" - team: "sre" - service: "vmagent-dataengg-prd-fb" - env: "prd" - priority: "p0" - type: "vmagent-fb" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dataengg-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dataengg-prd-fb.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -# Modify resources -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - requests: - cpu: 2 - memory: 2Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-agent-startree/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-agent-startree/custom-values.yaml deleted file mode 100644 index fb2df11..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-agent-startree/custom-values.yaml +++ /dev/null @@ -1,296 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-dataengg-startree-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-dataengg-startree-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-vmagent-startree-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-dataengg.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-dataengg-startree-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vmagent-dataengg-startree-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dataengg-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dataengg-startree-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 10 - memory: 32Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent-tmp" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-tmp" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-agent/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index 8b3b1db..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,296 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-dataengg-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-dataengg-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-deng-sre-vmagnt-prd-mds@meesho-dataengg-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-dataengg.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-dataengg-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vmagent-dataengg-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dataengg-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dataengg-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 12 - memory: 24Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent-mds" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-mds" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml deleted file mode 100644 index 49ec3a5..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert-dr - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-dataengg-prd-dr - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/dataengg/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-dataengg-prd-dr.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-dataengg-prd-dr" - env: "prd" - priority: "p0" - type: "vmalert-dr" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - dedicated: "vmagent-dr" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert-startree-stateful/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert-startree-stateful/custom-values.yaml deleted file mode 100644 index 7d6458d..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert-startree-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-dataengg-startree-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-dataengg-startree-prd-proxy.victoriametrics-startree.svc.cluster.local:8481/select/0/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-dataengg-startree-prd.victoriametrics-startree.svc.cluster.local:8480/insert/0/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-dataengg-startree-prd-proxy.victoriametrics-startree.svc.cluster.local:8481/select/0/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/pinot/data-intelligence/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-dataengg-startree-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-dataengg-startree-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmstack-startree" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstack-startree" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert-startree/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert-startree/custom-values.yaml deleted file mode 100644 index de91417..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert-startree/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-dataengg-startree-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-dataengg-startree-prd-proxy.victoriametrics-startree.svc.cluster.local:8481/select/0/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-dataengg-startree-prd.victoriametrics-startree.svc.cluster.local:8480/insert/0/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-dataengg-startree-prd-proxy.victoriametrics-startree.svc.cluster.local:8481/select/0/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/pinot/data-intelligence/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-dataengg-startree-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-dataengg-startree-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - dedicated: "vmstack-startree" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstack-startree" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml deleted file mode 100644 index 8d9b106..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-dataengg-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-dataengg-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-dataengg-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-dataengg-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/dataengg/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-dataengg-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "dataengg" - team: "sre" - service: "vmalert-dataengg-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert/custom-values.yaml deleted file mode 100644 index cbe5d74..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-alert/custom-values.yaml +++ /dev/null @@ -1,305 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-dataengg-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-dataengg-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-dataengg-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-dataengg-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config1/dataengg/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-dataengg-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-dataengg-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-insert-startree/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-insert-startree/custom-values.yaml deleted file mode 100644 index 75ef952..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-insert-startree/custom-values.yaml +++ /dev/null @@ -1,234 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dataengg-startree-prd - replicaCount: 3 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-dataengg-startree-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 50 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vminsert-dataengg-startree-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 2 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 60 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstack-startree" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstack-startree" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 2 - memory: 1.6Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-dataengg-startree-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: nginx-internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-insert/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index a5328e2..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,234 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dataengg-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-dataengg-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vminsert-dataengg-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 4 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 60 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vminsert-mds" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 10 - memory: 10Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-dataengg-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: contour-internal-0 - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-select-startree/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-select-startree/custom-values.yaml deleted file mode 100644 index c128f5a..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-select-startree/custom-values.yaml +++ /dev/null @@ -1,293 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dataengg-startree-prd - replicaCount: 3 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-dataengg-startree-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 180s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 96 - clusternative.maxConcurrentRequests: 256 - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vmselect-dataengg-startree-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 25 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "vmstack-startree" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstack-startree" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 5 - memory: 18Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-dataengg-startree-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-select/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index 8eb4599..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,293 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dataengg-prd - replicaCount: 8 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-dataengg-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 180s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 150 - clusternative.maxConcurrentRequests: 256 - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vmselect-dataengg-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 25 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 5 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "vmselect-mds" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 17 - memory: 31Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: contour-internal-0 - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-dataengg.prd.meesho.int - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-storage-startree/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-storage-startree/custom-values.yaml deleted file mode 100644 index f658795..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-storage-startree/custom-values.yaml +++ /dev/null @@ -1,316 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-dataengg-startree-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstack-startree" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstack-startree" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 500Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vmstorage-dataengg-startree-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 3 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 6 - memory: 16Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.93.7-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-dataengg-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-storage/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index 4a5bf62..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,316 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-dataengg-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 700Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vmstorage-dataengg-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 12 - memory: 160Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.93.7-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-dataengg-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoriametrics-agent/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index c0e28e7..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,485 +0,0 @@ -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "dataengg-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -# -- Replica count -replicaCount: 2 - -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - -# -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables -deployment: - spec: - minReadySeconds: 1200 - progressDeadlineSeconds: 1800 - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - -image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - # -- Image tag, set to `Chart.AppVersion` by default - tag: v1.133.0 # rewrites Chart.AppVersion - # -- Variant of the image to use. - # e.g. enterprise, scratch - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - -# -- Image pull secrets -imagePullSecrets: [] - -# -- Add additional DNS entries to pods hosts file. Check [official documentation](https://kubernetes.io/docs/tasks/network/customize-hosts-file-for-pods/) -hostAliases: [] -# - ip: 192.168.1.1 -# hostNames: -# - test.example.com -# - another.example.net - -# -- Override chart name -nameOverride: "" - -# -- Override resources fullname -fullnameOverride: "vm-agent-dataengg-prd" - -# -- Container working directory -containerWorkingDir: "/" - -rbac: - # -- Enables Role/RoleBinding creation - create: true - - # -- Role/RoleBinding annotations - annotations: {} - - # -- Role/RoleBinding labels - extraLabels: {} - - # -- If true and `rbac.enabled`, will deploy a Role/RoleBinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - - # -- additional rules for a role - extraRules: [] - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - # -- Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-deng-sre-vmagnt-prd-mds@meesho-dataengg-prd-0622.iam.gserviceaccount.com - } - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # -- mount API token to pod directly - automountToken: true - -# -- See `kubectl explain poddisruptionbudget.spec` for more or check [official documentation](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# -- Generates `remoteWrite.*` flags and config maps with value content for values, that are of type list of map. -# Each item should contain `url` param to pass validation. -remoteWrite: - - url: http://vm-insert-dataengg-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write - disableOnDiskQueue: false - dropSamplesOnOverload: false - - url: http://prd-census-server-dataengg.prd-census-server-dataengg.svc.cluster.local/api/v1/write - disableOnDiskQueue: true - dropSamplesOnOverload: true -# urlRelabelConfig: -# - action: keep -# source_labels: [env] -# regex: "dev" - -# -- VMAgent extra command line arguments -extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8429 - promscrape.config.strictParse: false # default is true - promscrape.maxScrapeSize: 1000000000 # default is 167772160 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - # promscrape.suppressScrapeErrors: true # default is false - - # Uncomment and specify the port if you want to support any of the protocols: - # https://docs.victoriametrics.com/victoriametrics/vmagent/#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for more details. -env: - - name: GOGC - value: "200" - - name: DRAGONFLY_BEARER_TOKEN - valueFrom: - secretKeyRef: - key: token - name: dragonfly-cloud-creds - -# -- Specify alternative source for env variables -envFrom: - [] - #- configMapRef: - # name: special-config - -# -- Extra labels for Deployment and Statefulset -extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vm-agent-dataengg-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra labels for Pods only -podLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vm-agent-dataengg-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra selector labels common for pod and service -selectorLabels: {} - -# -- Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# -- Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# -- Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -# -- Extra containers to run in a pod with vmagent -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -# -- Init containers for vmagent -initContainers: - [] - # - name: example - # image: example-image - -# -- Security context to be added to pod -podSecurityContext: - enabled: true - # fsGroup: 2000 - -# -- Security context to be added to pod's containers -securityContext: - enabled: true - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - # -- Enable agent service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - extraLabels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) for details - externalIPs: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8429 - # -- Target port - targetPort: http - # nodePort: 30000 - # -- Service type - type: ClusterIP - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - -ingress: - # -- Enable deployment of ingress for agent - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-agent-dataengg-prd.meeshogcp.in - path: - - / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - - # -- Ingress controller class name - ingressClassName: "nginx-internal" - - # -- Ingress path type - pathType: Prefix - -# -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) -resources: - requests: - cpu: 25 - memory: 24Gi - -# -- Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) -nodeSelector: - dedicated: "vmagent-mds" - -# -- Node tolerations for server scheduling to nodes with taints. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-mds" - effect: "NoSchedule" - -# -- Pod topologySpreadConstraints -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# -- Pod affinity -affinity: {} - -# -- VMAgent [scraping configuration](https://docs.victoriametrics.com/victoriametrics/vmagent/#how-to-collect-metrics-in-prometheus-format) -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-dataengg-prd" - -# -- Priority class to be assigned to the pod(s) -priorityClassName: "" - -# -- Enable the host network -hostNetwork: false - -serviceMonitor: - # -- Enable deployment of Service Monitor for server component. This is Prometheus operator object - enabled: false - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Service Monitor relabelings - relabelings: [] - # -- Basic auth params for Service Monitor - basicAuth: {} - # -- Service Monitor metricRelabelings - metricRelabelings: [] - # -- Service Monitor targetPort - targetPort: http - # interval: 15s - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - -# -- Empty dir configuration for a case, when persistence is disabled -emptyDir: {} - -persistentVolume: - # -- Create/use Persistent Volume Claim for server component. Empty dir if false - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Size of the volume. Should be calculated based on the logs you send and retention policy you set. - size: 10Gi - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume additional labels - extraLabels: {} - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Horizontal Pod Autoscaling. -# Note that it is not intended to be used for vmagents which perform scraping. -# In order to scale scraping vmagents check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) -horizontalPodAutoscaling: - # -- Use HPA for vmagent - enabled: false - # -- Maximum replicas for HPA to use to to scale vmagent - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale vmagent - minReplicas: 1 - # -- Metric for HPA to use to scale vmagent - metrics: [] - -# -- VMAgent scrape configuration -config: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -probe: - # -- Readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - # -- Liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - # -- Startup probe - startup: {} - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -allowedMetricsEndpoints: - - /metrics - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoriametrics-insert/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoriametrics-insert/custom-values.yaml deleted file mode 100644 index 67452f0..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoriametrics-insert/custom-values.yaml +++ /dev/null @@ -1,411 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: false - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- IDs of vmstorage nodes to exclude from writing - excludeStorageIDs: [] - # -- Override default `app` label name - name: vminsert - # -- VMInsert strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - # override Chart.AppVersion - tag: v1.107.0-cluster - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMInsert http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-dataengg-prd - # -- Extra command line arguments for vminsert component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8480 - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-dataengg-prd-0.vm-storage-dataengg-prd.victoriametrics.svc:8400" - - "vm-storage-dataengg-prd-1.vm-storage-dataengg-prd.victoriametrics.svc:8400" - - "vm-storage-dataengg-prd-2.vm-storage-dataengg-prd.victoriametrics.svc:8400" - - "vm-storage-dataengg-prd-3.vm-storage-dataengg-prd.victoriametrics.svc:8400" - - "vm-storage-dataengg-prd-4.vm-storage-dataengg-prd.victoriametrics.svc:8400" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vm-insert-dataengg-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Pod's additional labels - podLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vm-insert-dataengg-prd" - env: "prd" - priority: "p0" - type: "vminsert" - terminationGracePeriodSeconds: 30 - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - - # -- Readiness & Liveness probes - probe: - # -- VMInsert readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert startup probe - startup: {} - - # -- Relabel configuration - relabel: - enabled: false - config: [] - # -- Use existing configmap if specified - # otherwise .config values will be used. Relabel config **should** reside under `relabel.yml` key - configMap: "" - - # Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 4 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 60 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vminsert - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vminsert - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - dedicated: "vminsert-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 10 - memory: 10Gi - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - service: - # -- Create VMInsert service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here]( https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) for details - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-insert-dataengg-prd.meeshogcp.in - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - - # -- Ingress controller class name - ingressClassName: nginx-internal - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for insert component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/insert" .Values.vminsert }}' - - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vminsert component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoriametrics-select/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoriametrics-select/custom-values.yaml deleted file mode 100644 index 3abc78c..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoriametrics-select/custom-values.yaml +++ /dev/null @@ -1,447 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - splitService: true - externalService: - enabled: true - name: vmselect-dataengg-prd-proxy - # -- Override default `app` label name - clusternativeService: - enabled: false - name: "" - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # override Chart.AppVersion - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMSelect http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-dataengg-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - # -- Extra command line arguments for vmselect component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8481 - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 180s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-dataengg-prd-0.vm-storage-dataengg-prd.victoriametrics.svc:8401" - - "vm-storage-dataengg-prd-1.vm-storage-dataengg-prd.victoriametrics.svc:8401" - - "vm-storage-dataengg-prd-2.vm-storage-dataengg-prd.victoriametrics.svc:8401" - - "vm-storage-dataengg-prd-3.vm-storage-dataengg-prd.victoriametrics.svc:8401" - - "vm-storage-dataengg-prd-4.vm-storage-dataengg-prd.victoriametrics.svc:8401" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vm-select-dataengg-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # -- Pod's additional labels - podLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vm-select-dataengg-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Readiness & Liveness probes - probe: - # -- VMSelect readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 25 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 5 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vmselect - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vmselect - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - dedicated: "vmselect-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 17 - memory: 31Gi - - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Create VMSelect service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-select-dataengg.prd.meesho.int - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # -- Ingress controller class name - ingressClassName: contour-internal-0 - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for select component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/select" .Values.vmselect }}' - - # -- vmselect mode: deployment, daemonSet - mode: deployment - - # -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables - deployment: - spec: - # -- VMSelect strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - - # -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables - statefulSet: - enabled: false - spec: - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume extra labels - extraLabels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmselect component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: false - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - diff --git a/helm-overrides/k8s-dataengg-prd-ase1/victoriametrics-storage/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1/victoriametrics-storage/custom-values.yaml deleted file mode 100644 index 8dd4e55..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1/victoriametrics-storage/custom-values.yaml +++ /dev/null @@ -1,363 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - enabled: false - -vminsert: - enabled: false - -vmauth: - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- Override default `app` label name - name: vmstorage - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Variant of the image to use. e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMStorage http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-dataengg-prd - - # -- Additional environment variables (ex.: secret tokens, flags). Check https://docs.victoriametrics.com/victoriametrics/#environment-variables for details - env: [] - # -- Specify alternative source for env variables - envFrom: [] - # -- Data retention period. Possible units character: h(ours), d(ays), w(eeks), y(ears), if no unit character specified - month. The minimum retention period is 24h. - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - httpListenAddr: :8482 - loggerTimezone: "Asia/Kolkata" - dedup.minScrapeInterval: 60s - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage-n4d" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage-n4d" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - # -- Use an alternate scheduler, e.g. "stork". Check https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ for details - # schedulerName: - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Override Persistent Volume Claim name - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are https://kubernetes.io/docs/concepts/storage/persistent-volumes/ - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume extra labels - extraLabels: {} - # -- Storage class name. Will be empty if not set - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume - size: 1300Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vm-storage-dataengg-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - # -- Pod's additional labels - podLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vm-storage-dataengg-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Details are https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 26 - memory: 220Gi - - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: - enabled: false - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - podSecurityContext: - enabled: false - service: - enabled: true - # -- Service annotations - annotations: {} - # -- Service ClusterIP - clusterIP: None - # -- Service type - type: ClusterIP - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmselect - vmselectPort: 8401 - # -- Extra service ports - extraPorts: [] - # -- Health check node port for a service. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilyPolicy: "" - # -- List of service IP families. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilies: [] - # -- Traffic Distribution. Check https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution - trafficDistribution: "" - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - minReadySeconds: 5 - # -- Readiness probes - probe: - # -- VMStorage readiness probe - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMStorage startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmstorage component - enabled: false - # -- Maximum replicas for HPA to use to to scale the vmstorage component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmstorage component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmstorage component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: - scaleDown: - selectPolicy: Disabled - - vmbackupmanager: - # -- Enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" \ No newline at end of file diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/conntrack-adjuster/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 7b64b15..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - # Maximum number of conntrack entries - max: 2097152 - # Hash size for conntrack - hashsize: 524288 - # Sleep interval between adjustments (seconds) - sleepInterval: 30 - -# Enable additional matchExpressions -# addExtraMatchExpressions: true - -# Additional matchExpressions to append -# additionalMatchExpressions: -# - key: node-role -# operator: In -# values: -# - "worker" -# - "ingress" -# - key: environment -# operator: In -# values: -# - "production" -# - "staging" \ No newline at end of file diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/contour-cert-checker/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/contour-cert-checker/custom-values.yaml deleted file mode 100644 index e73bb2a..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=k8s-dataengg-prd-ase1c"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - dedicated: dataengg-devops - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dataengg-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/contour-external/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/contour-external/custom-values.yaml deleted file mode 100644 index 5d76c3a..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/contour-external/custom-values.yaml +++ /dev/null @@ -1,139 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 2048Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external - nodeSelector: - dedicated: contour-external - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external - nodeSelector: - dedicated: contour-external - logLevel: error - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "50" - keda: - enabled: true - scaledown: - policies: - - periodseconds: 180 - type: Pods - value: 2 - selectpolicy: Min - stabilizationWindowSeconds: 1200 - scaleup: - policies: - - periodseconds: 15 - type: Percent - value: 20 - selectpolicy: Max - stabilizationWindowSeconds: 120 - # triggers: - # - metadata: - # desiredReplicas: "8" - # end: 50 8 * * * - # start: 15 7 * * * - # timezone: Asia/Kolkata - # type: cron - # - metadata: - # desiredReplicas: "12" - # end: 45 10 * * * - # start: 10 9 * * * - # timezone: Asia/Kolkata - # type: cron - # - metadata: - # desiredReplicas: "15" - # end: 45 10 * * * - # start: 10 10 * * * - # timezone: Asia/Kolkata - # type: cron - # - metadata: - # desiredReplicas: "18" - # end: 10 12 * * * - # start: 0 11 * * * - # timezone: Asia/Kolkata - # type: cron - # - metadata: - # desiredReplicas: "18" - # end: 55 15 * * * - # start: 45 14 * * * - # timezone: Asia/Kolkata - # type: cron - # - metadata: - # desiredReplicas: "20" - # end: 10 17 * * * - # start: 30 16 * * * - # timezone: Asia/Kolkata - # type: cron - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 8Gi - service: - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-dengg-c-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/contour-internal-0/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/contour-internal-0/custom-values.yaml deleted file mode 100644 index dee8897..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,91 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - image: - tag: 1.28.0-debian-11-r8 - enabled: true - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "55" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-dengg-c-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/contour-internal-1/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/contour-internal-1/custom-values.yaml deleted file mode 100644 index 5f26b28..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,94 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 1 - memory: 3Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - logLevel: error - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 300 - targetCPU: "50" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 12 - memory: 8Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int1-dengg-c-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/contour-internal-intra-0/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index 8ff560a..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,89 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - image: - tag: 1.28.0-debian-11-r8 - enabled: true - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 5 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "55" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/contour-internal-intra-1/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index 7013115..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,92 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 1 - memory: 3Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: dataengg - team: dataengg-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 300 - targetCPU: "50" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 12 - memory: 8Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/coredns/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/coredns/custom-values.yaml deleted file mode 100644 index 526cde6..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/coredns/custom-values.yaml +++ /dev/null @@ -1,30 +0,0 @@ -replicaCount: 16 - -labels: - bu: dataengg - team: dataengg-devops - env: prd - -clusterIP: 10.219.16.2 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "dataengg-devops" - effect: "NoSchedule" - -nodeSelector: - dedicated: dataengg-devops - kubernetes.io/os: linux - -overwriteRegion: "c" -communicationType: "" diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/external-secrets/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/external-secrets/custom-values.yaml deleted file mode 100644 index 8debf00..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/external-secrets/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Additional service annotations - annotations: {} - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dataengg-devops - nodeSelector: - dedicated: dataengg-devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dataengg-devops - nodeSelector: - dedicated: dataengg-devops - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dataengg-devops - nodeSelector: - dedicated: dataengg-devops diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/flagger/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/flagger/custom-values.yaml deleted file mode 100644 index c766dea..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/flagger/custom-values.yaml +++ /dev/null @@ -1,69 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - dedicated: dataengg-devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dataengg-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: dataengg-devops - bu: infra - region: ase1c - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-dataengg-rollout-service.prd-dataengg-rollout-service.svc.cluster.local" - - -rolloutService: - enabled: true - apps: - prd-devops-node-app-consumers: true - prd-devops-node-app: true - prd-devops-golang-app: true - prd-devops-golang-app-test: true - prd-devops-golang-app-crons: true diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/fluentd/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/fluentd/custom-values.yaml deleted file mode 100644 index afed364..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/fluentd/custom-values.yaml +++ /dev/null @@ -1,720 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-deng-desre-fluentd-prd@meesho-dataengg-ase1c-prd-0625.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: dataengg - team: sre - type: fluentd - service: fluentd-dataengg-prd - priority: p0 - env: prd - zone_extended: ase1c - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: dataengg - team: sre - type: fluentd - service: fluentd-dataengg-prd - priority: p0 - env: prd - zone_extended: ase1c - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: [] -# - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - # - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/ingress-nginx-external/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/ingress-nginx-external/custom-values.yaml deleted file mode 100644 index a22c75b..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/ingress-nginx-external/custom-values.yaml +++ /dev/null @@ -1,34 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - resources: - requests: - cpu: 100m - memory: 768Mi - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 15 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 70 - ingressClass: nginx-external - ingressClassByName: true - watchIngressWithoutClass: false - ingressClassResource: - controllerValue: k8s.io/ingress-nginx-external - name: nginx-external - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dataengg-external-prd"}}}' - nodeSelector: - dedicated: nginx-external - tolerations: - - key: "dedicated" - operator: "Equal" - value: "nginx-external" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/ingress-nginx-internal/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/ingress-nginx-internal/custom-values.yaml deleted file mode 100644 index 9a859df..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/ingress-nginx-internal/custom-values.yaml +++ /dev/null @@ -1,33 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - resources: - requests: - cpu: 100m - memory: 768Mi - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 45 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 70 - extraArgs: - enable-ssl-passthrough: true - tcp-services-configmap: nginx-dataengg-internal/tcp-services - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dataengg-internal-ase1c-prd"}}}' - nodeSelector: - dedicated: nginx-internal - tolerations: - - key: "dedicated" - operator: "Equal" - value: "nginx-internal" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/ingress-nginx/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/ingress-nginx/custom-values.yaml deleted file mode 100644 index 685d900..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/ingress-nginx/custom-values.yaml +++ /dev/null @@ -1,32 +0,0 @@ -ingress-nginx: - controller: - config: - proxy-body-size: "50g" - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - resources: - requests: - cpu: 200m - memory: 512Mi - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 30 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dataengg-ase1c-prd"}}}' - nodeSelector: - dedicated: devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/keda/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/keda/custom-values.yaml deleted file mode 100644 index d2139af..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/keda/custom-values.yaml +++ /dev/null @@ -1,61 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: dataengg-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "dataengg-devops" - effect: "NoSchedule" - podLabels: - bu: "dataengg" - team: "dataengg-devops" - metricsAdapter: - bu: "dataengg" - team: "dataengg-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 2000Mi - requests: - cpu: 250m - memory: 300Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 30m - memory: 100Mi - metricServer: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 100m - memory: 200Mi - podAnnotations: - # -- Pod annotations for KEDA operator - keda: - prometheus.io/path: /metrics - prometheus.io/port: '8080' - prometheus.io/scrape: 'true' - # -- Pod annotations for KEDA Metrics Adapter - metricsAdapter: - prometheus.io/path: /metrics - prometheus.io/port: '8080' - prometheus.io/scrape: 'true' - # -- Pod annotations for KEDA Admission webhooks - webhooks: - prometheus.io/path: /metrics - prometheus.io/port: '8080' - prometheus.io/scrape: 'true' - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/kube-dns/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/kube-dns/custom-values.yaml deleted file mode 100644 index e6fe725..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.219.16.2"],"prd.mrouter.int.svc.cluster.local":["10.219.16.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.219.16.2"]} diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/kube-events/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/kube-events/custom-values.yaml deleted file mode 100644 index a12fea0..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-dataengg-ase1c-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - dedicated: dataengg-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: dataengg-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: dataengg-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: dataengg-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: dataengg-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: dataengg-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/kube-state-metrics/custom-values.yaml deleted file mode 100644 index e3de251..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-dataengg-ase1c-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-deng-ase1c-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "kube-state-metrics-dataengg-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "vmselect" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 60m - memory: 800Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index b3f5fa8..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,119 +0,0 @@ -fullnameOverride: opentelemetry-dataengg-ase1c-prd - -mode: daemonset - -priorityClassName: "system-node-critical" - -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-central-prd-ase1c-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external - - key: node_pool - operator: NotIn - values: - - np-dataengg-default-prd-ase1 - -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 500m - memory: 256Mi - limits: - cpu: 1 - memory: 512Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "dataengg" - team: "sre" - service: "opentelemetry-dataengg-ase1c-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - zone_extended: "ase1c" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index 9774b24..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,498 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-admin-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - - -# Extra labels to be added to node exporter pods -podLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "node-exporter-dataengg-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - zone_extended: "ase1c" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index 71d9472..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-dataengg-ase1c-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-dataengg-ase1c-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "stackdriver-exporter-dataengg-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - zone_extended: "ase1c" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-dataengg-ase1c-prd-0225" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,compute.googleapis.com/instance,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - dedicated: "vmselect" - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-deng-desre-stackdriver-prd@meesho-dataengg-ase1c-prd-0622.iam.gserviceaccount.com - } - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/telegraf-operator/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/telegraf-operator/custom-values.yaml deleted file mode 100644 index 11b7a3d..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,151 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "dataengg" - team: "dataengg-sre" - service: "telegraf-operator-dataengg-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - zone_extended: "ase1c" - -nodeSelector: - dedicated: "vmselect" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/victoria-metrics-agent/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index 8b3242f..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,298 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-dataengg-ase1c-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-dataengg-ase1c-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-common-np-deng-prd@meesho-dataengg-ase1c-prd-0625.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - http://vminsert-dataengg-ase1c-prd.deng-c.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-dataengg-ase1c-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write - # - http://vminsert-dataengg-ase1c-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vmagent-dataengg-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmagent" - zone_extended: "ase1c" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dataengg-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dataengg-ase1c-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 10 - memory: 32Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/victoria-metrics-insert/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index 95f7402..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,235 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dataengg-ase1c-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-dataengg-ase1c-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vminsert-dataengg-ase1c-prd" - env: "prd" - priority: "p0" - type: "vminsert" - zone_extended: "ase1c" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 1 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 60 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vminsert" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 3 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 4 - memory: 2Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-dataengg-ase1c-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/victoria-metrics-select/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index 4294742..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,301 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dataengg-ase1c-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-dataengg-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - clusternative.maxConcurrentRequests: "256" - dedup.minScrapeInterval: 60s - search.maxConcurrentRequests: "150" - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 180s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vmselect-dataengg-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmselect" - zone_extended: "ase1c" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 15 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 1 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "vmselect" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 1 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 9 - memory: 16Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-dataengg-ase1c-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - clusternativeService: - enabled: true - port: 8401 - sessionAffinity: None - ipFamilyPolicy: SingleStack - - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dataengg-prd-ase1c/victoria-metrics-storage/custom-values.yaml b/helm-overrides/k8s-dataengg-prd-ase1c/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index 4637dcb..0000000 --- a/helm-overrides/k8s-dataengg-prd-ase1c/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,317 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-dataengg-ase1c-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 500Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "dataengg" - team: "dataengg-sre" - service: "vmstorage-dataengg-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - zone_extended: "ase1c" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 5 - memory: 70Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.93.7-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-dataengg-ase1c-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-datascience-prd-ase1/README.md b/helm-overrides/k8s-datascience-prd-ase1/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/k8s-datascience-prd-ase1/alloy/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/alloy/custom-values.yaml deleted file mode 100644 index f110f24..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/alloy/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -fullnameOverride: "alloy-datascience-prd" - -alloy: - configMap: - configFile: datascience.alloy - - clustering: - enabled: true - - extraPorts: - - name: "otlp-grpc" - port: 4317 - targetPort: 4317 - protocol: "TCP" - - name: "otlp-http" - port: 4318 - targetPort: 4318 - protocol: "TCP" - - resources: - requests: - cpu: 7 - memory: 55Gi - -configReloader: - enabled: true - -serviceAccount: - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-sre-grafna-obs-stk-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - -controller: - # Must be one of 'daemonset', 'deployment', or 'statefulset'. - type: 'deployment' - - nodeSelector: - dedicated: "alloy" - tolerations: - - key: "dedicated" - operator: "Equal" - value: "alloy" - effect: "NoSchedule" - - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 50 - targetCPUUtilizationPercentage: 80 - targetMemoryUtilizationPercentage: 80 \ No newline at end of file diff --git a/helm-overrides/k8s-datascience-prd-ase1/aurva-dataplane/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/aurva-dataplane/custom-values.yaml deleted file mode 100644 index bb6b546..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/aurva-dataplane/custom-values.yaml +++ /dev/null @@ -1,675 +0,0 @@ -## Aurva Data Plane -## Ref: https://github.com/aurva-io/aurva-charts.git - -postgresql: - enabled: true - fullnameOverride: "aurva-dataplane-database" - volumePermissions: - ## @param volumePermissions.enabled Enable init container that changes the owner and group of the persistent volume - ## - enabled: true - global: - storageClass: pd-standard-retain - postgresql: - auth: - postgresPassword: "aurva" - database: "controller" - # Add toleration to make sure where this postgres db pod should reside (Applicable for production workloads): For more detail ref: https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/ - primary: - extendedConfiguration: | - max_connections = 300 - - # #PLACEHOLDER## - # tolerations: - # - effect: NoSchedule - # key: dedicated - # operator: Equal - # value: datascience-devops - - # # -- Select nodes to deploy which matches the following labels - # nodeSelector: ##PLACEHOLDER## - # dedicated: datascience-devops - -# -- Provide a name in place of `aurva` -# namespaceOverride: aurva-dataplane - -########################################################## -# Global Configs -########################################################## -global: - aurva_controller: - enabled: true - aurva_fastdet: - enabled: false - aurva_pii_analyzer: - enabled: true - aurva_ocr: - enabled: true - aurva_collector: - enabled: true - - deploymentAnnotations: {} - - priorityClassName: "" - -########################################################## -# Aurva Controller -########################################################## -aurva_controller: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "datascience" - team: "datascience-devops" - service: "aurva-datascience-prd" - env: "prd" - priority: "p0" - type: "aurva_controller" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 10 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - image: - # Image of the app container - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-controller/aurva-controller - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - - type: secret - name: aurva-controller-secrets - # - type: configmap - # name: proxy-datasource-config - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 2 - requests: - memory: 1Gi - cpu: 1 - - aurvaFastdet: - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-fastdet/aurva-fastdet - tag: "v2.30.13" - pullPolicy: IfNotPresent - envFrom: [] - env: [] - resources: - limits: - cpu: 0.5 - memory: 512Mi - requests: - cpu: 0.5 - memory: 512Mi - - nodeSelector: ##PLACEHOLDER## - dedicated: datascience-devops - - # -- Taint tolerations for nodes - ##PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: datascience-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-controller-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - config: - #variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - COMMAND_URL: "command.aurva-prd.meeshogcp.in:80" - DEPLOYMENT_TYPE: "kubernetes" - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - FLUSHER_WORKER_POOL_SIZE: "1000" - FLUSHER_BATCH_SIZE: "30000" - UNIQUENESS_IDENTIFIER: "k8s-datascience-prd-ase1" #Recommendation: should be equal to cluster name - PROVIDER_ACCOUNT_ID: "meesho-datascience-prd-0622" # GCP PROJECT ID (not Number) - REGION: "asia-southeast1" #eg: asia-south1 - ENVIRONMENT: "prod" - OCR_ENABLED: "true" - MONITORING_ENABLED: "false" - HYBRID_ONLY_MODE: "false" - FORCE_TLS: "false" - FASTDET_FLAG : "true" - AADHAAR_ENHANCER: "0" - WORKSPACE_EVENT_TRACKING_ENABLED: "false" - ACCESS_IQ_ENABLED: "true" - GRPC_ENFORCE_ALPN_ENABLED: "false" - ENABLE_FASTDET: "true" - FASTDET_MAX_BATCH_SIZE: "100" - HEARTBEAT_INTERVAL: "5m" - #pii data - PII_BUCKET_NAME: "gcs-infra-devop-aurva-datascience-prd" - PII_LOG_BUCKET_REGION: "asia-southeast1" - PII_LOG_CRON: "*/5 * * * *" - ENABLE_PII_LOG: "true" - PII_EVIDENCE_UPLOAD_MAX_BLOCKING_TASKS: "50000" - PII_EVIDENCE_UPLOAD_MAX_CONCURRENT_TASKS: "500" - PII_EVIDENCE_MAX_CACHE_WEIGHT: "100" - QUOTA_CLEANUP_CRON: "0 0 * * *" - ENABLE_PII_QUOTA: "true" - MAX_PII_EVIDENCES_PER_KEY_PER_WINDOW: "3" - PII_QUOTA_SYNC_CRON: "*/2 * * * *" - QUOTA_WINDOW_HOURS: "24" - #constants - SKIP_NAMESPACES: "contour-internal-1-datascience-prd,contour-internal-0-datascience-prd,contour-external-datascience-prd" - CLOUD_PROVIDER: "gcp" - LOG_ENV: "production" - RDS_SCANNER_AVAILABILITY : "false" - REDSHIFT_SCANNER_AVAILABILITY : "false" - S3_SCANNER_AVAILABILITY : "false" - DYNAMO_SCANNER_AVAILABILITY: "false" - DOCDB_SCANNER_AVAILABILITY: "false" - OPENSEARCH_SCANNER_AVAILABILITY: "false" - CLOUDSQL_SCANNER_AVAILABILITY: "true" - BIGQUERY_SCANNER_AVAILABILITY: "true" - AWS_SNAPSHOT_SCANNER_AVAILABILITY: "false" - CLOUDSTORAGE_SCANNER_AVAILABILITY: "true" - KEYSPACES_SCANNER_AVAILABILITY: "false" - ALLOYDB_SCANNER_AVAILABILITY: "true" - BIGTABLE_SCANNER_AVAILABILITY: "true" - GCP_BACKUP_AVAILABILITY: "true" - EGRESS_MODE_ONLY: "false" - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - SCAN_UUID_ENABLED: "true" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-controller-sa - # -- Annotations applied to created service account - annotations: - iam.gke.io/gcp-service-account: sa-datascience-prd-aurva-contr@meesho-datascience-prd-0622.iam.gserviceaccount.com -# eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 15 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 - - -########################################################## -# Aurva OCR -########################################################## -aurva_ocr: - # -- Additional labels for aurva-controller - additionalLabels: - bu: "datascience" - team: "datascience-devops" - service: "aurva-datascience-prd" - env: "prd" - priority: "p0" - type: "aurva_ocr" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 1 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-ocr/aurva-ocr - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-ocr: - type: secret - name: aurva-ocr-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 1 - requests: - memory: 2Gi - cpu: 1 - - # -- Select nodes to deploy which matches the following labels - - nodeSelector: ##PLACEHOLDER## - dedicated: datascience-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: datascience-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-ocr-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - OCR_TIME_LIMIT: "1" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-datascience-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-ocr-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # iam.gke.io/gcp-service-account: service-account@gcp.iam.gserviceaccount.com - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva Collector -########################################################## -aurva_collector: - - # -- Additional labels for aurva-analyzer - additionalLabels: - bu: "datascience" - team: "datascience-devops" - service: "aurva-datascience-prd" - env: "prd" - priority: "p0" - type: "aurva_collector" - - - # -- Annotations on aurva-analyzer - annotations: {} -# "key": "value" - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-collector/aurva-collector - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-controller: - type: secret - name: aurva-collector-secrets - - resources: - limits: - cpu: 800m - memory: 800Mi - requests: - cpu: 200m - memory: 512Mi - - podSecurityContext: {} - - securityContext: - privileged: true - capabilities: - add: - # For kernel v5.8 and above we don't need CAP_SYS_ADMIN or CAP_SYS_RESOURCE - # we just need CAP_BPF and CAP_PERFMON. This has been tested on our EKS node - # which is on kernel v5.10.x - # When SSL Tracing is required we need CAP_SYS_ADMIN and CAP_SYS_PTRACE - # on top of the previous capabilities - # So finally these are the 4 possible combinations for capabilies - # 1. Newer Kernels without SSL - # - BPF - # - PERFMON - # 2. Newer Kernels with SSL - - SYS_ADMIN - - SYS_PTRACE - # 3. Older Kernels without SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # 4. Older Kernels with SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # - SYS_PTRACE - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - volumes: - - name: debugfs - mountPath: /sys/kernel/debug - hostPath: /sys/kernel/debug - - name: vmlinux - mountPath: /sys/kernel/btf/vmlinux - hostPath: /sys/kernel/btf/vmlinux - - name: procfs - mountPath: /host/proc - hostPath: /proc - - name: bpffs - mountPath: /sys/fs/bpf - hostPath: /sys/fs/bpf - - # -- Taint tolerations for nodes - tolerations: - # - effect: NoSchedule - # key: dedicated - # operator: Equal - # value: megatetra - - operator: Exists - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external - - alloy - - preprod-spot-16 - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-collector-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - # variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-datascience-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - TRACE_INTERNAL_SVC: "true" - TRACE_INTERNAL_SVC_HTTP: "true" - INTERNAL_SVC_SAMPLE_INTERVAL: "10m" - LOGS_TTL: "1h" - TRACE_HTTP2: "true" - TRACE_SSL: "false" - TRACE_PSQL: "false" - TRACE_SQLSERVER: "false" - TRACE_MYSQL: "false" - TRACE_EGRESS: "true" - TRACE_GO_TLS: "false" - TRACE_ML_SERVICES: "false" - # constants - LOG_ENV: production - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - MONITORING_ENABLED: "false" - ENABLE_INGRESS_INFORMER: "false" - ENABLE_SERVICE_INFORMER: "false" - ENABLE_ISTIO_INFORMER: "false" - EXCLUDED_PII_REGEX_TYPES: "ip_address,us_bank_number,us_driver_license,us_itin,us_passport,us_routing,us_mbi,ssn" - AGGREGATOR_MAX_CONNECTIONS: "1000" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-collector-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva PII Analyzer -########################################################## -aurva_pii_analyzer: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "datascience" - team: "datascience-devops" - service: "aurva-datascience-prd" - env: "prd" - priority: "p0" - type: "aurva_pii_analyzer" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 3 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - ##PLACEHOLDER## - nodeSelector: - dedicated: datascience-devops - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-piianalyzer/aurva-piianalyzer - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-pii-analyzer: - type: secret - name: aurva-pii-analyzer-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 4Gi - cpu: 4 - requests: - memory: 2Gi - cpu: 2 - - nodeSelector: ##PLACEHOLDER## - dedicated: datascience-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: datascience-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-pii-analyzer-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - SCHEDULER_TIME: "1" - SUPPORTED_REGION: "US" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-datascience-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-pii-analyzer-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 3 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 diff --git a/helm-overrides/k8s-datascience-prd-ase1/cert-manager/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/cert-manager/custom-values.yaml deleted file mode 100644 index b555322..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/cert-manager/custom-values.yaml +++ /dev/null @@ -1,129 +0,0 @@ -global: - logLevel: 2 - rbac: - create: true - priorityClassName: "high-priority" -installCRDs: false - -crds: - enabled: true - keep: true - -# Cert-manager Controller -replicaCount: 1 -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-controller - tag: v1.20.1 - pullPolicy: IfNotPresent - -nodeSelector: - dedicated: datascience-devops - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - -resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 500m - memory: 512Mi - -# Webhook Configuration -webhook: - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-webhook - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: datascience-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# CA Injector Configuration -cainjector: - enabled: true - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-cainjector - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: datascience-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# ACME Solver Configuration -acmesolver: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-acmesolver - tag: v1.20.1 - pullPolicy: IfNotPresent - -# Startup API Check -startupapicheck: - enabled: true - timeout: 1m - backoffLimit: 4 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-startupapicheck - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: datascience-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - -# Prometheus Monitoring -prometheus: - enabled: true - servicemonitor: - enabled: false - interval: 60s - scrapeTimeout: 30s - labels: - prometheus: cert-manager diff --git a/helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-0-arm.yaml b/helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-0-arm.yaml deleted file mode 100644 index 02536e5..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-0-arm.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-0-arm -spec: - nodePoolConfig: - serviceAccount: sa-dsci-shared-nodepool-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-8 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-1-arm.yaml b/helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-1-arm.yaml deleted file mode 100644 index 39c7cd5..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-1-arm.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-1-arm -spec: - nodePoolConfig: - serviceAccount: sa-dsci-shared-nodepool-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-dataproc-arm.yaml b/helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-dataproc-arm.yaml deleted file mode 100644 index 102aa2e..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-dataproc-arm.yaml +++ /dev/null @@ -1,22 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-dataproc-arm -spec: - nodePoolConfig: - serviceAccount: sa-dsci-shared-nodepool-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-intra-0-arm.yaml b/helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-intra-0-arm.yaml deleted file mode 100644 index e58159c..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-intra-0-arm.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-intra-0-arm -spec: - nodePoolConfig: - serviceAccount: sa-dsci-shared-nodepool-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-intra-1-arm.yaml b/helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-intra-1-arm.yaml deleted file mode 100644 index e2a8b27..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-internal-intra-1-arm.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-internal-intra-1-arm -spec: - nodePoolConfig: - serviceAccount: sa-dsci-shared-nodepool-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-32 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-32 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-shared-arm.yaml b/helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-shared-arm.yaml deleted file mode 100644 index 80d7420..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/computeclass/contour-shared-arm.yaml +++ /dev/null @@ -1,28 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: contour-shared-arm -spec: - nodePoolConfig: - serviceAccount: sa-dsci-shared-nodepool-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - priorities: - - machineType: n4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - - machineType: c4d-highcpu-16 - maxPodsPerNode: 16 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: hyperdisk-balanced - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-datascience-prd-ase1/computeclass/g2-standard-16-l4-compute-class.yaml b/helm-overrides/k8s-datascience-prd-ase1/computeclass/g2-standard-16-l4-compute-class.yaml deleted file mode 100644 index c0aef9f..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/computeclass/g2-standard-16-l4-compute-class.yaml +++ /dev/null @@ -1,49 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-16-l4-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-a - machineType: g2-standard-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-b - machineType: g2-standard-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-c - machineType: g2-standard-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-datascience-prd-ase1/computeclass/g2-standard-4-l4-compute-class.yaml b/helm-overrides/k8s-datascience-prd-ase1/computeclass/g2-standard-4-l4-compute-class.yaml deleted file mode 100644 index 4fc4e9f..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/computeclass/g2-standard-4-l4-compute-class.yaml +++ /dev/null @@ -1,49 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-4-l4-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-a - machineType: g2-standard-4 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-b - machineType: g2-standard-4 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-c - machineType: g2-standard-4 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-datascience-prd-ase1/computeclass/g2-standard-8-l4-300gb-compute-class.yaml b/helm-overrides/k8s-datascience-prd-ase1/computeclass/g2-standard-8-l4-300gb-compute-class.yaml deleted file mode 100644 index 1910745..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/computeclass/g2-standard-8-l4-300gb-compute-class.yaml +++ /dev/null @@ -1,52 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-8-l4-300gb-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-a - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskSize: 300 - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-b - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskSize: 300 - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-c - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskSize: 300 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-datascience-prd-ase1/computeclass/g2-standard-8-l4-compute-class.yaml b/helm-overrides/k8s-datascience-prd-ase1/computeclass/g2-standard-8-l4-compute-class.yaml deleted file mode 100644 index ca5d26e..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/computeclass/g2-standard-8-l4-compute-class.yaml +++ /dev/null @@ -1,49 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-8-l4-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-a - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-b - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-c - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-datascience-prd-ase1/conntrack-adjuster/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 8dbbd40..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,20 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - max: 2097152 - hashsize: 524288 - sleepInterval: 30 - -nodeAffinity: - enabled: true - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: In - values: - - "contour-internal-0" - - "contour-internal-1-c4d" - - "contour-internal-0-c4d" - - "contour-intra-1" - - "megaquad" diff --git a/helm-overrides/k8s-datascience-prd-ase1/contour-ca-issuer/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/contour-ca-issuer/custom-values.yaml deleted file mode 100644 index 5e5d014..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/contour-ca-issuer/custom-values.yaml +++ /dev/null @@ -1,17 +0,0 @@ -# ClusterIssuer override for k8s-datascience-prd-ase1 (prd datascience cluster, ase1a zone). -# ClusterIssuer is a cluster-scoped resource — this file pins the issuer -# identity for this cluster so the name is auditable per-cluster. -# -# Must match the `issuerRef.name` in consuming Certificate CRs (see -# devops-helm-charts/2.0.0/templates/proxyless-grpc-cert.yaml). - -issuerName: contour-datascience-prd-ca-issuer -rootCASecretName: contour-datascience-ca - -externalSecret: - enabled: true - vaultPath: meesho/devops/contour/root-ca - refreshInterval: "0" - secretStoreRef: - name: vault-backend -namespace: cert-manager-datascience-prd \ No newline at end of file diff --git a/helm-overrides/k8s-datascience-prd-ase1/contour-cert-checker/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/contour-cert-checker/custom-values.yaml deleted file mode 100644 index 599fe15..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=k8s-datascience-prd-ase1"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - dedicated: datascience-devops - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: datascience-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/k8s-datascience-prd-ase1/contour-internal-0/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/contour-internal-0/custom-values.yaml deleted file mode 100644 index fedd5dc..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,117 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 60s - max-connection-duration: 600s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 500m - memory: 3Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - tcpLB: true - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 5% - maxUnavailable: 0 - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-0-arm - nodeSelector: - cloud.google.com/compute-class: contour-internal-0-arm - logLevel: error - extraArgs: - - '--concurrency 6' - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 10Gi - limits: - cpu: 6 - memory: 12Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-dsci-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-datascience-prd-ase1/contour-internal-1/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/contour-internal-1/custom-values.yaml deleted file mode 100644 index b4d3d26..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,118 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 400s - max-connection-duration: 540s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 8 - memory: 10Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - tcpLB: true - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-1-arm - nodeSelector: - cloud.google.com/compute-class: contour-internal-1-arm - logLevel: error - extraArgs: - - '--concurrency 14' - terminationGracePeriodSeconds: 500 - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 500 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 600 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 50 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "25" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 14 - memory: 8Gi - limits: - cpu: 14 - memory: 27Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int1-dsci-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-datascience-prd-ase1/contour-internal-dataproc/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/contour-internal-dataproc/custom-values.yaml deleted file mode 100644 index fbbf78a..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/contour-internal-dataproc/custom-values.yaml +++ /dev/null @@ -1,112 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 60s - max-connection-duration: 600s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 500m - memory: 3Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-dataproc" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 5% - maxUnavailable: 0 - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-dataproc-arm - nodeSelector: - cloud.google.com/compute-class: contour-internal-dataproc-arm - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 14' - resources: - requests: - cpu: 14 - memory: 8Gi - limits: - cpu: 14 - memory: 27Gi - service: - tcpLB: true - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-datascience-prd-ase1/contour-internal-intra-0/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index 2d11141..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,114 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 60s - max-connection-duration: 600s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 500m - memory: 3Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 5% - maxUnavailable: 0 - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-intra-0-arm - nodeSelector: - cloud.google.com/compute-class: contour-internal-intra-0-arm - logLevel: error - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 14' - resources: - requests: - cpu: 14 - memory: 8Gi - limits: - cpu: 14 - memory: 27Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-datascience-prd-ase1/contour-internal-intra-1/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index 89871cd..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,115 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 400s - max-connection-duration: 540s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 8 - memory: 10Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-shared-arm - nodeSelector: - cloud.google.com/compute-class: contour-shared-arm - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: contour-internal-intra-1-arm - nodeSelector: - cloud.google.com/compute-class: contour-internal-intra-1-arm - logLevel: error - terminationGracePeriodSeconds: 500 - autoscaling: - behavior: - scaleDown: - policies: - - periodSeconds: 60 - type: Pods - value: 2 - selectPolicy: Min - stabilizationWindowSeconds: 300 - scaleUp: - policies: - - periodSeconds: 15 - type: Percent - value: 100 - selectPolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - extraArgs: - - '--concurrency 20' - resources: - requests: - cpu: 20 - memory: 12Gi - limits: - cpu: 30 - memory: 57Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-datascience-prd-ase1/coredns/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/coredns/custom-values.yaml deleted file mode 100644 index 02dba43..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/coredns/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -replicaCount: 40 -communicationType: "intra" - -labels: - bu: datascience - team: datascience-devops - env: prd - -clusterIP: 10.137.16.2 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - -nodeSelector: - dedicated: datascience-devops - kubernetes.io/os: linux diff --git a/helm-overrides/k8s-datascience-prd-ase1/coroot-node-agent/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/coroot-node-agent/custom-values.yaml deleted file mode 100644 index 18798eb..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/coroot-node-agent/custom-values.yaml +++ /dev/null @@ -1,52 +0,0 @@ -fullnameOverride: "coroot-node-agent-datascience-prd" -priorityClassName: "low-priority" - -scrape: "false" # disable scraping for now - -resources: - requests: - cpu: "300m" - memory: "1Gi" - limits: - cpu: "500m" - memory: "1.5Gi" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-datascience-contour-ext - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-datascience-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external \ No newline at end of file diff --git a/helm-overrides/k8s-datascience-prd-ase1/external-secrets/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/external-secrets/custom-values.yaml deleted file mode 100644 index d0ce0d1..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/external-secrets/custom-values.yaml +++ /dev/null @@ -1,61 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Annotations to add to Pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: datascience-devops - nodeSelector: - dedicated: datascience-devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: datascience-devops - nodeSelector: - dedicated: datascience-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: datascience-devops - nodeSelector: - dedicated: datascience-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" \ No newline at end of file diff --git a/helm-overrides/k8s-datascience-prd-ase1/flagger/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/flagger/custom-values.yaml deleted file mode 100644 index 6405568..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/flagger/custom-values.yaml +++ /dev/null @@ -1,62 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - dedicated: datascience-devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: datascience-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: datascience-devops - bu: infra - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-datascience-rollout-service.prd-datascience-rollout-service.svc.cluster.local" - -rolloutService: - enabled: true - apps: diff --git a/helm-overrides/k8s-datascience-prd-ase1/fluentd/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/fluentd/custom-values.yaml deleted file mode 100644 index a561d05..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/fluentd/custom-values.yaml +++ /dev/null @@ -1,731 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-dssre-fluentd-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: datascience - team: sre - type: fluentd - service: fluentd-datascience-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: datascience - team: sre - type: fluentd - service: fluentd-datascience-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: - - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - from_encoding "utf-8" - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-datascience-prd-ase1/ingress-nginx-internal/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/ingress-nginx-internal/custom-values.yaml deleted file mode 100644 index 4192db9..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/ingress-nginx-internal/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 6 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dsci-int-prd"}}}' - nodeSelector: - dedicated: nginx-internal - tolerations: - - key: "dedicated" - operator: "Equal" - value: "nginx-internal" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-datascience-prd-ase1/keda/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/keda/custom-values.yaml deleted file mode 100644 index 44e99a2..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/keda/custom-values.yaml +++ /dev/null @@ -1,46 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: datascience-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - podLabels: - bu: "datascience" - team: "datascience-devops" - metricsAdapter: - bu: "datascience" - team: "datascience-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 500m - memory: 450Mi - # -- Manage [resource request & limits] of KEDA metrics apiserver pod - metricServer: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 300m - memory: 200Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 100m - memory: 180Mi - requests: - cpu: 60m - memory: 120Mi - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' diff --git a/helm-overrides/k8s-datascience-prd-ase1/kube-dns/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/kube-dns/custom-values.yaml deleted file mode 100644 index 4563673..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.137.16.2"],"prd.mrouter.int.svc.cluster.local":["10.137.16.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.137.16.2"]} diff --git a/helm-overrides/k8s-datascience-prd-ase1/kube-events/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/kube-events/custom-values.yaml deleted file mode 100644 index 0c332bb..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-datascience-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - dedicated: datascience-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: datascience-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: datascience-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: datascience-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: datascience-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: datascience-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-datascience-prd-ase1/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/kube-state-metrics/custom-values.yaml deleted file mode 100644 index c2f59bf..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-datascience-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-datascience-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "datascience" - team: "datascience-sre" - service: "kube-state-metrics-datascience-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "vmselect" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 17m - memory: 1.5Gi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-datascience-prd-ase1/kubectl-mcp-server/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/kubectl-mcp-server/custom-values.yaml deleted file mode 100644 index 53c110e..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/kubectl-mcp-server/custom-values.yaml +++ /dev/null @@ -1,117 +0,0 @@ -fullnameOverride: "kubectl-mcp-server" - -replicas: 1 - -image: - # In-house hardened build (helm-templates/kubectl-mcp-server/docker/) — - # NOT the upstream Docker Hub image. 266→18 HIGH/CRIT vulns, non-root, - # multi-stage slim base, kubectl v1.33.12 / helm v3.21.0. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/prd/devop/kubectl-mcp-server - tag: "v2" - pullPolicy: IfNotPresent - -labels: - bu: infra - team: devops - service: kubectl-mcp-server - env: prd - -serviceAccount: - create: true - annotations: {} - -rbac: - create: true - -podAnnotations: {} - -podSecurityContext: {} - -securityContext: - runAsNonRoot: true - runAsUser: 1000 - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -priorityClassName: "" - -nodeSelector: - dedicated: "datascience-devops" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "datascience-devops" - effect: NoSchedule - -affinity: {} - -resources: - requests: - cpu: 250m - memory: 512Mi - limits: - cpu: 1000m - memory: 1Gi - -mcp: - mode: single - transport: http - host: "0.0.0.0" - port: 8000 - -auth: - allowAnonymous: false - -externalSecrets: - enabled: true - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/prd/cntr/devop/kubectl-mcp-server-datascience" - -# tcpSocket on purpose — streamable-http transport exposes only /mcp, -# there is no /health route (see chart values.yaml note). -livenessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - -readinessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - -service: - port: 8000 - type: ClusterIP - -# Contour standard (2.0.0-style): renders HTTPProxy parent/child + -intra -# instead of a networking.k8s.io Ingress. -createContourGateway: true - -ingress: - enabled: true - ingressClassName: contour-internal-1 - servicePortNumber: 8000 - hosts: - - host: kubectl-mcp-server-datascience.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 diff --git a/helm-overrides/k8s-datascience-prd-ase1/kubernetes-dashboard/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/kubernetes-dashboard/custom-values.yaml deleted file mode 100644 index cd5b892..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/kubernetes-dashboard/custom-values.yaml +++ /dev/null @@ -1,442 +0,0 @@ -# Copyright 2017 The Kubernetes Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -# General configuration shared across resources -app: - # Mode determines if chart should deploy a full Dashboard with all containers or just the API. - # - dashboard - deploys all the containers - # - api - deploys just the API - mode: 'dashboard' - image: - pullPolicy: IfNotPresent - pullSecrets: [] - scheduling: - # Node labels for pod assignment - # Ref: https://kubernetes.io/docs/user-guide/node-selection/ - nodeSelector: {} - security: - # Allow overriding csrfKey used by API/Auth containers. - # It has to be base64 encoded random 256 bytes string. - # If empty, it will be autogenerated. - csrfKey: ~ - # SecurityContext to be added to pods - # To disable set the following configuration to null: - # securityContext: null - securityContext: - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - # ContainerSecurityContext to be added to containers - # To disable set the following configuration to null: - # containerSecurityContext: null - containerSecurityContext: - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - runAsUser: 1001 - runAsGroup: 2001 - capabilities: - drop: ["ALL"] - # Pod Disruption Budget configuration - # Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - minAvailable: 0 - maxUnavailable: 0 - networkPolicy: - enabled: false - ingressDenyAll: false - # Raw network policy spec that overrides predefined spec - # Example: - # spec: - # egress: - # - ports: - # - port: 123 - spec: {} - - # Common labels & annotations shared across all deployed resources - labels: {} - annotations: {} - # Common priority class used for all deployed resources - priorityClassName: null - settings: - ## Global dashboard settings - global: - # # Cluster name that appears in the browser window title if it is set - # clusterName: "" - # # Max number of items that can be displayed on each list page - # itemsPerPage: 10 - # # Max number of labels that are displayed by default on most views. - # labelsLimit: 3 - # # Number of seconds between every auto-refresh of logs - # logsAutoRefreshTimeInterval: 5 - # # Number of seconds between every auto-refresh of every resource. Set 0 to disable - # resourceAutoRefreshTimeInterval: 10 - # # Hide all access denied warnings in the notification panel - # disableAccessDeniedNotifications: false - # # Hide all namespaces option in namespace selection dropdown to avoid accidental selection in large clusters thus preventing OOM errors - # hideAllNamespaces: false - # # Namespace that should be selected by default after logging in. - # defaultNamespace: default - # # List of namespaces that should be presented to user without namespace list privileges. - # namespaceFallbackList: - # - default - ## Pinned resources that will be displayed in dashboard's menu - pinnedResources: [] - # - kind: customresourcedefinition - # # Fully qualified name of a CRD - # name: prometheus.monitoring.coreos.com - # # Display name - # displayName: Prometheus - # # Is this CRD namespaced? - # namespaced: true - ingress: - enabled: false - hosts: - # Keep 'localhost' host only if you want to access Dashboard using 'kubectl port-forward ...' on: - # https://localhost:8443 - - localhost - # - kubernetes.dashboard.domain.com - ingressClassName: internal-nginx - # Use only if your ingress controllers support default ingress classes. - # If set to true ingressClassName will be ignored and not added to the Ingress resources. - # It should fall back to using IngressClass marked as the default. - useDefaultIngressClass: false - # This will append our Ingress with annotations required by our default configuration. - # nginx.ingress.kubernetes.io/backend-protocol: "HTTPS" - # nginx.ingress.kubernetes.io/ssl-passthrough: "true" - # nginx.ingress.kubernetes.io/ssl-redirect: "true" - useDefaultAnnotations: true - pathType: ImplementationSpecific - # If path is not the default (/), rewrite-target annotation will be added to the Ingress. - # It allows serving Kubernetes Dashboard on a sub-path. Make sure that the configured path - # does not conflict with gateway route configuration. - path: / - issuer: - name: selfsigned - # Scope determines what kind of issuer annotation will be used on ingress resource - # - default - adds 'cert-manager.io/issuer' - # - cluster - adds 'cert-manager.io/cluster-issuer' - # - disabled - disables cert-manager annotations - scope: default - tls: - enabled: true - # If provided it will override autogenerated secret name - secretName: "" - labels: {} - annotations: {} - # Use the following toleration if Dashboard can be deployed on a tainted control-plane nodes - # - key: node-role.kubernetes.io/control-plane - # effect: NoSchedule - tolerations: [] - affinity: {} - -auth: - role: auth - image: - repository: docker.io/kubernetesui/dashboard-auth - tag: 1.2.2 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: auth - containerPort: 8000 - protocol: TCP - args: [] - env: [] - volumeMounts: - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for Auth related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -# API deployment configuration -api: - role: api - image: - repository: docker.io/kubernetesui/dashboard-api - tag: 1.10.1 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: api - containerPort: 8000 - protocol: TCP - # Additional container arguments - # Full list of arguments: https://github.com/kubernetes/dashboard/blob/master/docs/common/arguments.md - # args: - # - --system-banner="Welcome to the Kubernetes Dashboard" - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store exec logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for API related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -# WEB UI deployment configuration -web: - role: web - image: - repository: docker.io/kubernetesui/dashboard-web - tag: 1.6.0 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: web - containerPort: 8000 - protocol: TCP - # Additional container arguments - # Full list of arguments: https://github.com/kubernetes/dashboard/blob/master/docs/common/arguments.md - # args: - # - --system-banner="Welcome to the Kubernetes Dashboard" - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for WEB UI related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -### Metrics Scraper -### Container to scrape, store, and retrieve a window of time from the Metrics Server. -### refs: https://github.com/kubernetes/dashboard/tree/master/modules/metrics-scraper -metricsScraper: - enabled: true - role: metrics-scraper - image: - repository: docker.io/kubernetesui/dashboard-metrics-scraper - tag: 1.2.1 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - containerPort: 8000 - protocol: TCP - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - livenessProbe: - httpGet: - scheme: HTTP - path: / - port: 8000 - initialDelaySeconds: 30 - timeoutSeconds: 30 - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for Metrics Scraper related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -## Optional Metrics Server sub-chart configuration -## Enable this if you don't already have metrics-server enabled on your cluster and -## want to use it with dashboard metrics-scraper -## refs: -## - https://github.com/kubernetes-sigs/metrics-server -## - https://github.com/kubernetes-sigs/metrics-server/tree/master/charts/metrics-server -metrics-server: - enabled: false - args: - - --kubelet-preferred-address-types=InternalIP - - --kubelet-insecure-tls - -## Required Kong sub-chart with DBless configuration to act as a gateway -## for our all containers. -kong: - enabled: false - admin: - tls: - enabled: false - ## Configuration reference: https://docs.konghq.com/gateway/3.6.x/reference/configuration - env: - dns_order: LAST,A,CNAME,AAAA,SRV - plugins: 'off' - nginx_worker_processes: 1 - ingressController: - enabled: false - manager: - enabled: false - dblessConfig: - configMap: kong-dbless-config - proxy: - type: ClusterIP - http: - enabled: true - -## Optional Cert Manager sub-chart configuration -## Enable this if you don't already have cert-manager enabled on your cluster. -cert-manager: - enabled: false - installCRDs: true - -## Optional Nginx Ingress sub-chart configuration -## Enable this if you don't already have nginx-ingress enabled on your cluster. -nginx: - enabled: false - controller: - electionID: ingress-controller-leader - ingressClassResource: - name: internal-nginx - default: false - controllerValue: k8s.io/internal-ingress-nginx - service: - type: ClusterIP - -## Extra configurations: -## - manifests -## - predefined roles -## - prometheus -## - etc... -extras: - # Extra Kubernetes manifests to be deployed - # manifests: - # - apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: additional-configmap - # data: - # mykey: myvalue - manifests: [] - serviceMonitor: - # Whether to create a Prometheus Operator service monitor. - enabled: false - # Here labels can be added to the serviceMonitor - labels: {} - # Here annotations can be added to the serviceMonitor - annotations: {} - # metrics.serviceMonitor.metricRelabelings Specify Metric Relabelings to add to the scrape endpoint - # ref: https://github.com/coreos/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # metrics.serviceMonitor.relabelings [array] Prometheus relabeling rules - relabelings: [] - # ServiceMonitor connection scheme. Defaults to HTTPS. - scheme: https - # ServiceMonitor connection tlsConfig. Defaults to {insecureSkipVerify:true}. - tlsConfig: - insecureSkipVerify: true diff --git a/helm-overrides/k8s-datascience-prd-ase1/kyverno/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/kyverno/custom-values.yaml deleted file mode 100644 index e7593a2..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/kyverno/custom-values.yaml +++ /dev/null @@ -1,2240 +0,0 @@ -global: - - # -- Internal settings used with `helm template` to generate install manifest - # @ignored - templating: - enabled: false - debug: false - version: ~ - - image: - # -- (string) Global value that allows to set a single image registry across all deployments. - # When set, it will override any values set under `.image.registry` across the chart. - registry: ~ - # -- (list) Global list of Image pull secrets - # When set, it will override any values set under `imagePullSecrets` under different components across the chart. - imagePullSecrets: [] - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - caCertificates: - # -- Global CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - # Individual controller values will override this global value - data: ~ - - # -- Global value to set single volume to be mounted for CA certificates for all deployments. - # Not used when `.Values.global.caCertificates.data` is defined - # Individual controller values will override this global value - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Additional container environment variables to apply to all containers and init containers - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Global node labels for pod assignment. Non-global values will override the global value. - nodeSelector: - dedicated: datascience-kyverno - - # -- Global List of node taints to tolerate. Non-global values will override the global value. - tolerations: - - key: "dedicated" - operator: "Equal" - value: "datascience-kyverno" - effect: "NoSchedule" - -# -- (string) Override the name of the chart -nameOverride: ~ - -# -- (string) Override the expanded name of the chart -fullnameOverride: ~ - -# -- (string) Override the namespace the chart deploys to -namespaceOverride: ~ - -upgrade: - # -- Upgrading from v2 to v3 is not allowed by default, set this to true once changes have been reviewed. - fromV2: false - -apiVersionOverride: - # -- (string) Override api version used to create `PodDisruptionBudget`` resources. - # When not specified the chart will check if `policy/v1/PodDisruptionBudget` is available to - # determine the api version automatically. - podDisruptionBudget: ~ - -rbac: - roles: - # -- Aggregate ClusterRoles to Kubernetes default user-facing roles. For more information, see [User-facing roles](https://kubernetes.io/docs/reference/access-authn-authz/rbac/#user-facing-roles) - aggregate: - admin: true - view: true - -# Use openreports.io as the API group for reporting -openreports: - # -- Enable OpenReports feature in controllers - enabled: false - # -- Whether to install CRDs from the upstream OpenReports chart. Setting this to true requires enabled to also be true. - installCrds: false - -# CRDs configuration -crds: - - # -- Whether to have Helm install the Kyverno CRDs, if the CRDs are not installed by Helm, they must be added before policies can be created - install: true - - reportsServer: - # -- Kyverno reports-server is used in your cluster - enabled: false - - groups: - - # -- Install CRDs in group `kyverno.io` - kyverno: - cleanuppolicies: true - clustercleanuppolicies: true - clusterpolicies: true - globalcontextentries: true - policies: true - policyexceptions: true - updaterequests: true - - # -- Install CRDs in group `policies.kyverno.io` - policies: - validatingpolicies: true - policyexceptions: true - imagevalidatingpolicies: true - namespacedimagevalidatingpolicies: true - mutatingpolicies: true - generatingpolicies: true - deletingpolicies: true - namespaceddeletingpolicies: true - namespacedvalidatingpolicies: true - - # -- Install CRDs in group `reports.kyverno.io` - reports: - clusterephemeralreports: true - ephemeralreports: true - - # -- Install CRDs in group `wgpolicyk8s.io` - wgpolicyk8s: - clusterpolicyreports: false ##lk - policyreports: false ##lk - - # -- Additional CRDs annotations (Replace=true avoids 262144-byte annotation limit on large CRDs) - annotations: - argocd.argoproj.io/sync-options: Replace=true - # strategy.spinnaker.io/replace: 'true' - - # -- Additional CRDs labels - customLabels: {} - - migration: - - # -- Enable CRDs migration using helm post upgrade hook - enabled: true - - # -- Resources to migrate - resources: - - cleanuppolicies.kyverno.io - - clustercleanuppolicies.kyverno.io - - clusterpolicies.kyverno.io - - globalcontextentries.kyverno.io - - policies.kyverno.io - - policyexceptions.kyverno.io - - updaterequests.kyverno.io - - deletingpolicies.policies.kyverno.io - - generatingpolicies.policies.kyverno.io - - imagevalidatingpolicies.policies.kyverno.io - - namespacedimagevalidatingpolicies.policies.kyverno.io - - mutatingpolicies.policies.kyverno.io - - namespaceddeletingpolicies.policies.kyverno.io - - namespacedvalidatingpolicies.policies.kyverno.io - - policyexceptions.policies.kyverno.io - - validatingpolicies.policies.kyverno.io - - image: - # -- (string) Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- (string) Image repository - repository: kyverno/kyverno-cli - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- (string) Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podResources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -# Configuration -config: - - # -- Create the configmap. - create: true - - # -- Preserve the configmap settings during upgrade. - preserve: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - # -- Enable registry mutation for container images. Enabled by default. - enableDefaultRegistryMutation: true - - # -- The registry hostname used for the image mutation. - defaultRegistry: docker.io - - # -- Exclude groups - excludeGroups: - - system:nodes - - # -- Exclude usernames - excludeUsernames: [] - # - '!system:kube-scheduler' - - # -- Exclude roles - excludeRoles: [] - - # -- Exclude roles - excludeClusterRoles: [] - - # -- Generate success events. - generateSuccessEvents: false - - # -- Resource types to be skipped by the Kyverno policy engine. - # Make sure to surround each entry in quotes so that it doesn't get parsed as a nested YAML list. - # These are joined together without spaces, run through `tpl`, and the result is set in the config map. - # @default -- See [values.yaml](values.yaml) - resourceFilters: - - '[Event,*,*]' - # - '[*/*,kube-system,*]' - - '[*/*,kube-public,*]' - - '[*/*,kube-node-lease,*]' - - '[Node,*,*]' - - '[Node/?*,*,*]' - - '[APIService,*,*]' - - '[APIService/?*,*,*]' - - '[TokenReview,*,*]' - - '[SubjectAccessReview,*,*]' - - '[SelfSubjectAccessReview,*,*]' - - '[Binding,*,*]' - - '[Pod/binding,*,*]' - - '[ReplicaSet,*,*]' - - '[ReplicaSet/?*,*,*]' - - '[EphemeralReport,*,*]' - - '[ClusterEphemeralReport,*,*]' - # exclude resources from the chart - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:additional]' - - '[ClusterRoleBinding,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.configMapName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.metricsConfigMapName" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Job,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[Job/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.admission-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.background-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - # -- Sets the threshold for the total number of UpdateRequests generated for mutateExisitng and generate policies. - updateRequestThreshold: 1000 - - # -- Defines the `namespaceSelector`/`objectSelector` in the webhook configurations. - # The Kyverno namespace is excluded if `excludeKyvernoNamespace` is `true` (default) - webhooks: - # Exclude namespaces - # namespaceSelector: lk-s2 - # matchExpressions: - # - key: kubernetes.io/metadata.name - # operator: NotIn - # values: - # - kube-systemm lk-s2 - # Exclude objects - # objectSelector: - # matchExpressions: - # - key: webhooks.kyverno.io/exclude - # operator: DoesNotExist - - # -- Defines annotations to set on webhook configurations. - webhookAnnotations: - # Example to disable admission enforcer on AKS: - # 'admissions.enforcer/disabled': 'true' lk-s3 - - # -- Defines labels to set on webhook configurations. - webhookLabels: {} - # Example to adopt webhook resources in ArgoCD: - # 'argocd.argoproj.io/instance': 'kyverno' - - # -- Defines match conditions to set on webhook configurations (requires Kubernetes 1.27+). - matchConditions: [] - - # -- Exclude Kyverno namespace - # Determines if default Kyverno namespace exclusion is enabled for webhooks and resourceFilters - excludeKyvernoNamespace: true - - # -- resourceFilter namespace exclude - # Namespaces to exclude from the default resourceFilters - resourceFiltersExcludeNamespaces: [] - - # -- resourceFilters exclude list - # Items to exclude from config.resourceFilters - resourceFiltersExclude: [] - - # -- resourceFilter namespace include - # Namespaces to include to the default resourceFilters - resourceFiltersIncludeNamespaces: [] - - # -- resourceFilters include list - # Items to include to config.resourceFilters - resourceFiltersInclude: [] - -# Metrics configuration -metricsConfig: - - # -- Create the configmap. - create: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - namespaces: - - # -- List of namespaces to capture metrics for. - include: [] - - # -- list of namespaces to NOT capture metrics for. - exclude: [] - - # -- (string) Rate at which metrics should reset so as to clean up the memory footprint of kyverno metrics, if you might be expecting high memory footprint of Kyverno's metrics. Default: 0, no refresh of metrics. WARNING: This flag is not working since Kyverno 1.8.0 - metricsRefreshInterval: ~ - # metricsRefreshInterval: 24h - - # -- (list) Configures the bucket boundaries for all Histogram metrics, changing this configuration requires restart of the kyverno admission controller - bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 15, 20, 25, 30] - - # -- (map) Configures the exposure of individual metrics, by default all metrics and all labels are exported, changing this configuration requires restart of the kyverno admission controller - metricsExposure: - kyverno_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_image_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_mutating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_generating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_admission_review_duration_seconds: - # enabled: false - disabledLabelDimensions: ["resource_namespace"] - kyverno_policy_rule_info_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_policy_results_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_admission_requests_total: - disabledLabelDimensions: ["resource_namespace"] - kyverno_cleanup_controller_deletedobjects_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - -# -- Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -imagePullSecrets: {} - # regcred: - # registry: foo.example.com - # username: foobar - # password: secret - # regcred2: - # registry: bar.example.com - # username: barbaz - # password: secret2 - -# -- Existing Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -existingImagePullSecrets: [] - # - test-registry - # - other-test-registry - -# Tests configuration -test: - # -- Sleep time before running test - sleep: 20 - - image: - # -- (string) Image registry - registry: curlimages - # -- Image repository - repository: curl - # -- Image tag - # Defaults to `latest` if omitted - tag: '8.10.1' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - # -- Security context for the test containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- Additional Pod annotations - podAnnotations: {} - - # -- List of node taints to tolerate - tolerations: [] - -# -- Additional labels -customLabels: {} - - -webhooksCleanup: - # -- Create a helm pre-delete hook to cleanup webhooks. - enabled: true - - autoDeleteWebhooks: - # -- Allow webhooks controller to delete webhooks using finalizers - enabled: false - - image: - # -- (string) Image registry - registry: registry.k8s.io - # -- Image repository - repository: kubectl - # -- Image tag - # Defaults to `latest` if omitted - tag: 'v1.32.7' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -grafana: - # -- Enable grafana dashboard creation. - enabled: false - - # -- Configmap name template. - configMapName: '{{ include "kyverno.fullname" . }}-grafana' - - # -- (string) Namespace to create the grafana dashboard configmap. - # If not set, it will be created in the same namespace where the chart is deployed. - namespace: ~ - - # -- Grafana dashboard configmap annotations. - annotations: {} - - # -- Grafana dashboard configmap labels - labels: - grafana_dashboard: "1" - - # -- create GrafanaDashboard custom resource referencing to the configMap. - # according to https://grafana-operator.github.io/grafana-operator/docs/examples/dashboard_from_configmap/readme/ - grafanaDashboard: - create: false - folder: kyverno - allowCrossNamespaceImport: true - matchLabels: - dashboards: "grafana" - -# Features configuration -features: - admissionReports: - # -- Enables the feature - enabled: true - aggregateReports: - # -- Enables the feature - enabled: true - policyReports: - # -- Enables the feature - enabled: false ##lk - validatingAdmissionPolicyReports: - # -- Enables the feature - enabled: true - mutatingAdmissionPolicyReports: - # -- Enables the feature - enabled: false - reporting: - # -- Enables the feature - validate: true - # -- Enables the feature - mutate: true - # -- Enables the feature - mutateExisting: true - # -- Enables the feature - imageVerify: true - # -- Enables the feature - generate: true - autoUpdateWebhooks: - # -- Enables the feature - enabled: true - backgroundScan: - # -- Enables the feature - enabled: true - # -- Number of background scan workers - backgroundScanWorkers: 2 - # -- Background scan interval - backgroundScanInterval: 1h - # -- Skips resource filters in background scan - skipResourceFilters: true - configMapCaching: - # -- Enables the feature - enabled: true - controllerRuntimeMetrics: - # -- Bind address for controller-runtime metrics (use "0" to disable it) - bindAddress: ":8080" - deferredLoading: - # -- Enables the feature - enabled: true - dumpPayload: - # -- Enables the feature - enabled: false - forceFailurePolicyIgnore: - # -- Enables the feature - enabled: false - generateValidatingAdmissionPolicy: - # -- Enables the feature - enabled: true - generateMutatingAdmissionPolicy: - # -- Enables the feature - enabled: false - dumpPatches: - # -- Enables the feature - enabled: false - globalContext: - # -- Maximum allowed response size from API Calls. A value of 0 bypasses checks (not recommended) - maxApiCallResponseLength: 2000000 - logging: - # -- Logging format - format: text - # -- Logging verbosity - verbosity: 2 - omitEvents: - # -- Events which should not be emitted (possible values `PolicyViolation`, `PolicyApplied`, `PolicyError`, and `PolicySkipped`) - eventTypes: - - PolicyApplied - - PolicySkipped - # - PolicyViolation - # - PolicyError - policyExceptions: - # -- Enables the feature - enabled: true - # -- Restrict policy exceptions to a single namespace - # Set to "*" to allow exceptions in all namespaces - namespace: '' - protectManagedResources: - # -- Enables the feature - enabled: false - registryClient: - # -- Allow insecure registry - allowInsecure: false - # -- Enable registry client helpers - credentialHelpers: - - default - - google - - amazon - - azure - - github - ttlController: - # -- Reconciliation interval for the label based cleanup manager - reconciliationInterval: 1m - tuf: - # -- Enables the feature - enabled: false - # -- (string) Path to Tuf root - root: ~ - # -- (string) Raw Tuf root - rootRaw: ~ - # -- (string) Tuf mirror - mirror: ~ - -# Admission controller configuration -admissionController: - autoscaling: - # -- Enable horizontal pod autoscaling - enabled: true - - # -- Minimum number of pods - minReplicas: 1 - - # -- Maximum number of pods - maxReplicas: 10 - - # -- Target CPU utilization percentage - targetCPUUtilizationPercentage: 80 - - # -- Configurable scaling behavior - behavior: {} - - # -- Overrides features defined at the root level - featuresOverride: - admissionReports: - # -- Max number of admission reports allowed in flight until the admission controller stops creating new ones - backPressureThreshold: 1000 - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- The ServiceAccount name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno admission controller activities. - # This will help ensure Kyverno stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- admissionController webhook server port - # in case you are using hostNetwork: true, you might want to change the port the webhookServer is listening to - webhookServer: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - admission-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.admissionController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - initContainer: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyvernopre - # -- (string) Image tag - # If missing, defaults to image.tag - tag: ~ - # -- (string) Image pull policy - # If missing, defaults to image.pullPolicy - pullPolicy: ~ - - resources: - # -- Pod resource limits - # limits: - # cpu: 100m - # memory: 256Mi - # -- Pod resource requests - requests: - cpu: 1000m - memory: 1Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - container: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyverno - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - resources: - # -- Pod resource limits - # limits: - # memory: 384Mi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Array of extra init containers - extraInitContainers: [] - # - name: init-container - # image: busybox - # command: ['sh', '-c', 'echo Hello'] - - # -- Array of extra containers to run alongside kyverno - extraContainers: [] - # - name: myapp-container - # image: busybox - # command: ['sh', '-c', 'echo Hello && sleep 3600'] - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Kyverno's metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Background controller configuration -backgroundController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable background controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: - - apiGroups: - - networking.k8s.io - resources: - - ingresses - - ingressclasses - - networkpolicies - verbs: - - create - - update - - patch - - delete - - apiGroups: - - rbac.authorization.k8s.io - resources: - - rolebindings - - roles - verbs: - - create - - update - - patch - - delete - - apiGroups: - - '' - resources: - - configmaps - - resourcequotas - - limitranges - verbs: - - create - - update - - patch - - delete - - apiGroups: - - resource.k8s.io - resources: - - resourceclaims - - resourceclaimtemplates - verbs: - - create - - delete - - update - - patch - - deletecollection - - apiGroups: - - autoscaling - resources: - - horizontalpodautoscalers - verbs: - - get - - list - - watch - - update - - patch - - apiGroups: - - apps - resources: - - deployments - - statefulsets - - replicasets - verbs: - - get - - list - - watch - - update - - patch - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - # - patch - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/background-controller - # -- Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - # limits: - # memory: 128Mi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - background-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.backgroundController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - # -- backgroundController server port - # in case you are using hostNetwork: true, you might want to change the port the backgroundController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Cleanup controller configuration -cleanupController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable cleanup controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - delete - # - list - # - watch - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/cleanup-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- cleanupController server port - # in case you are using hostNetwork: true, you might want to change the port the cleanupController is listening to - server: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - # limits: - # memory: 128Mi - # -- Pod resource requests - requests: - cpu: 1100m - memory: 1.2Gi - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - cleanup-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `service.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Reports controller configuration -reportsController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable reports controller. - enabled: false ##lk - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/reports-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno reports controller activities. - # This will help ensure Kyverno reports stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - memory: 128Mi - # -- Pod resource requests - requests: - cpu: 100m - memory: 64Mi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - reports-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.reportsController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- (string) Service node port. - # Only used if `type` is `NodePort`. - nodePort: ~ - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- (string) Traces receiver address - address: ~ - # -- (string) Traces receiver port - port: ~ - # -- (string) Traces receiver credentials - creds: ~ - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- (string) Otel collector endpoint - collector: ~ - # -- (string) Otel collector credentials - creds: ~ - - # -- reportsController server port - # in case you are using hostNetwork: true, you might want to change the port the reportsController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - - # -- Enable sanity check for reports CRDs - sanityChecks: true diff --git a/helm-overrides/k8s-datascience-prd-ase1/kyverno/policies/protect-namespace.yaml b/helm-overrides/k8s-datascience-prd-ase1/kyverno/policies/protect-namespace.yaml deleted file mode 100644 index 5db5729..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/kyverno/policies/protect-namespace.yaml +++ /dev/null @@ -1,40 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: protect-critical-namespaces -spec: - validationFailureAction: Enforce - background: false - rules: - - name: deny-deletion-of-critical-namespaces - match: - any: - - resources: - kinds: - - Namespace - names: - - keda-datascience-prd - - contour-internal-0-datascience-prd - - contour-internal-0-datascience-prd-intra - - contour-external-datascience-prd - - external-secrets-datascience-prd - - flagger-datascience-prd - - victoriametrics - - kube-system - - monitoring - - telegraf-operator - - loadtester - - opentelemetry - - fluentd - - kube-events - - observability - - contour-cert-checker-ns - - aurva-dataplane - validate: - message: "Deletion of critical namespaces is not allowed." - deny: - conditions: - any: - - key: "{{request.operation}}" - operator: Equals - value: DELETE diff --git a/helm-overrides/k8s-datascience-prd-ase1/kyverno/policies/restrict-replicas.yaml b/helm-overrides/k8s-datascience-prd-ase1/kyverno/policies/restrict-replicas.yaml deleted file mode 100644 index d456985..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/kyverno/policies/restrict-replicas.yaml +++ /dev/null @@ -1,29 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: restrict-minimum-replicas-selected-ns -spec: - validationFailureAction: Enforce - background: false - rules: - - name: min-replicas-deployments-statefulsets - match: - resources: - kinds: - - Deployment - - StatefulSet - - Deployment/scale - - StatefulSet/scale - names: - - coredns - namespaces: - - kube-system - operations: - - CREATE - - UPDATE - validate: - message: "Deployments and StatefulSets must have at least 5 replicas in this namespace." - anyPattern: - # Case 1: replicas explicitly set and >= 5 - - spec: - replicas: ">=5" diff --git a/helm-overrides/k8s-datascience-prd-ase1/loadtester/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/loadtester/custom-values.yaml deleted file mode 100644 index 76550f1..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/loadtester/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -replicaCount: 1 - -Namespace: loadtester - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/dev/devops/flagger-loadtester - tag: 0.30.0 - pullPolicy: IfNotPresent - pullSecret: - -podLabels: - bu: datascience - team: datascience-shared - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - -podPriorityClassName: "" - -logLevel: info -cmd: - timeout: 1h - namespaceRegexp: "" - -nameOverride: "loadtester" -fullnameOverride: "" - -env: [] - -service: - type: ClusterIP - port: 80 - -resources: - requests: - cpu: 10m - memory: 64Mi - -volumes: [] -volumeMounts: [] - -nodeSelector: - dedicated: datascience-devops - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - -affinity: {} - -ingress: - slowStart: - enabled: false - enabled: true - ingressClassName: contour-internal-1 - servicePort: http - hosts: - - host: datascience-flagger-loadtester.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / -createContourGateway: true -contourResponseTimeout: false -rbac: - # rbac.create: `true` if rbac resources should be created - create: true - # rbac.scope: `cluster` to create cluster-scope rbac resources (ClusterRole/ClusterRoleBinding) - # otherwise, namespace-scope rbac resources will be created (Role/RoleBinding) - scope: - # rbac.rules: array of rules to apply to the role. example: - # rules: - # - apiGroups: [""] - # resources: ["pods"] - # verbs: ["list", "get"] - rules: [] - -# name of an existing service account to use - if not creating rbac resources -serviceAccountName: "" - -# App Mesh virtual node settings (to be used for AppMesh v1beta1) -meshName: "" -#backends: -# - app1.namespace -# - app2.namespace - -# App Mesh virtual node settings (to be used for AppMesh v1beta2) -appmesh: - enabled: false - backends: - - podinfo - - podinfo-canary - -#Istio virtual service and gatway settings. TLS secrets should be in namespace before enbaled it. ( secret format loadtester.fullname ) -istio: - enabled: false - host: flagger-loadtester.flagger - gateway: - enabled: false - tls: - enabled: false - httpsRedirect: false - -# when enabled, it will add a security context for the loadtester pod -securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 100 - runAsGroup: 101 - -podDisruptionBudget: - enabled: false - minAvailable: 1 diff --git a/helm-overrides/k8s-datascience-prd-ase1/opentelemetry-coralogix/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/opentelemetry-coralogix/custom-values.yaml deleted file mode 100644 index 12794b1..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/opentelemetry-coralogix/custom-values.yaml +++ /dev/null @@ -1,281 +0,0 @@ -global: - domain: "coralogixsg.com" - defaultApplicationName: "default" - defaultSubsystemName: "nodes" - - # Old endpoint based configuration, - # please use domain instead. - traces: - endpoint: "" - metrics: - endpoint: "" - logs: - endpoint: "" - -# set distribution to openshift for openshift clusters -distribution: "" - -# Opentelemetry collector configuration -dedicatedValue: false -schedulerName: default-scheduler - -labels: - bu: "datascience" - team: "sre" - service: "opentelemetry-datascience-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: false - key: prd/admin/coralogix-keys - secretStoreRef: - name: vault-backend - -image: - # If you want to use the core image `otel/opentelemetry-collector`, you also need to change `command.name` value to `otelcol`. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - # repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - tag: "0.87.0" - # When digest is set to a non-empty value, images will be pulled by digest (regardless of tag value). - digest: "" -mode: daemonset -rollout: - rollingUpdate: - # When 'mode: daemonset', maxSurge cannot be used when hostPort is set for any of the ports - # maxSurge: 25% - maxUnavailable: 1 - strategy: RollingUpdate -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" -hostNetwork: true -dnsPolicy: "ClusterFirstWithHostNet" -fullnameOverride: opentelemetry-datascience-prd - -presets: - logsCollection: - enabled: false - storeCheckpoints: false - maxRecombineLogSize: 1048576 - extraFilelogOperators: [] -# - type: recombine -# combine_field: body -# source_identifier: attributes["log.file.path"] -# is_first_entry: body matches "^(YOUR-LOGS-REGEX)" - kubernetesAttributes: - enabled: false - hostMetrics: - enabled: false - kubeletMetrics: - enabled: false - -extraEnvs: -- name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" -- name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName -config: - extensions: {} - # zpages: - # endpoint: localhost:55679 - # pprof: - # endpoint: localhost:1777 - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 4s - tls: - insecure: true - sending_queue: - enabled: true - queue_size: 6000 - num_consumers: 200 - resolver: - dns: - hostname: opentelemetry-admin-prd.opentelemetry.svc.clusterset.local - # coralogix: - # timeout: "30s" - # private_key: "${CORALOGIX_PRIVATE_KEY}" - # domain: "{{.Values.global.domain}}" - # traces: - # endpoint: "{{ .Values.global.traces.endpoint }}" - # metrics: - # endpoint: "{{ .Values.global.metrics.endpoint }}" - # logs: - # endpoint: "{{ .Values.global.logs.endpoint }}" - # application_name_attributes: - # - "k8s.namespace.name" - # - "service.namespace" - # subsystem_name_attributes: - # - "k8s.deployment.name" - # - "k8s.statefulset.name" - # - "k8s.daemonset.name" - # - "k8s.cronjob.name" - # - "k8s.job.name" - # - "k8s.container.name" - # - "k8s.node.name" - # - "service.name" - # application_name: "{{.Values.global.defaultApplicationName }}" - # subsystem_name: "{{.Values.global.defaultSubsystemName }}" - processors: - k8sattributes: - filter: - node_from_env_var: KUBE_NODE_NAME - extract: - metadata: - - "k8s.namespace.name" - - "k8s.deployment.name" - - "k8s.statefulset.name" - - "k8s.daemonset.name" - - "k8s.cronjob.name" - - "k8s.job.name" - - "k8s.pod.name" - - "k8s.node.name" - memory_limiter: null # Will get the k8s resource limits - resourcedetection/env: - detectors: ["system","env"] - timeout: 2s - override: false - # spanmetrics: - # metrics_exporter: coralogix - # dimensions: - # - name: "k8s.deployment.name" - # - name: "k8s.statefulset.name" - # - name: "k8s.daemonset.name" - # - name: "k8s.cronjob.name" - # - name: "k8s.job.name" - # - name: "k8s.container.name" - # - name: "k8s.node.name" - # - name: "k8s.namespace.name" - receivers: - otlp: - protocols: - grpc: - endpoint: ${MY_POD_IP}:4317 - http: - endpoint: ${MY_POD_IP}:4318 - zipkin: - endpoint: ${MY_POD_IP}:9411 - jaeger: - protocols: - grpc: - endpoint: ${MY_POD_IP}:14250 - thrift_http: - endpoint: ${MY_POD_IP}:14268 - thrift_compact: - endpoint: ${MY_POD_IP}:6831 - thrift_binary: - endpoint: ${MY_POD_IP}:6832 - prometheus: - config: - scrape_configs: - - job_name: opentelemetry-collector - scrape_interval: 30s - static_configs: - - targets: - - ${MY_POD_IP}:8888 - service: - extensions: - # - zpages - # - pprof - - health_check -# - memory_ballast - telemetry: - metrics: - address: ${env:MY_POD_IP}:8888 - pipelines: - traces: - exporters: - - loadbalancing - processors: - - memory_limiter - # - spanmetrics - - batch - receivers: - - otlp - metrics: null - logs: null -tolerations: - - operator: Exists - -resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 1 - memory: 2G - -ports: - jaeger-binary: - enabled: true - containerPort: 6832 - servicePort: 6832 - hostPort: 6832 - protocol: TCP - # In order to enable podMonitor, following part must be enabled in order to expose the required port: - # metrics: - # enabled: true - -# podMonitor: -# enabled: true - -# prometheusRule: -# enabled: true -# defaultRules: -# enabled: true - -# Annotations to be added to pod -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-datascience-contour-ext - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-datascience-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert \ No newline at end of file diff --git a/helm-overrides/k8s-datascience-prd-ase1/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index 2bc1ef8..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,134 +0,0 @@ -fullnameOverride: opentelemetry-datascience-prd - -mode: daemonset - -# priorityClassName: "system-node-critical" -# Removing since dataengg pods are going into pending state -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-datascience-contour-ext - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-datascience-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external - -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 100m - memory: 124Mi - limits: - cpu: 200m - memory: 256Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "datascience" - team: "sre" - service: "opentelemetry-datascience-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/k8s-datascience-prd-ase1/paused-container/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/paused-container/custom-values.yaml deleted file mode 100644 index ffb4e28..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/paused-container/custom-values.yaml +++ /dev/null @@ -1,82 +0,0 @@ -# Default values for hello-world. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -priorityClass: - name: "low-priority" - -image: - repository: nginx - tag: "1.14.2" - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - -nameOverride: "" -fullnameOverride: "paused-container-datascience-prd" - -extraLabels: - team: "devops" - bu: "datascience" - env: "prd" - service: "paused-container-datascience-prd" - priority: "p3" - type: "tools" - -topologySpreadConstraints: [] - # - labelSelector: - # matchLabels: - # dedicated: vminsert - # maxSkew: 1 - # topologyKey: topology.kubernetes.io/hostname - # whenUnsatisfiable: DoNotSchedule - -affinity: - podAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - - labelSelector: - matchExpressions: - - key: type - operator: In - values: - - vmselect - - vminsert - - tools - topologyKey: topology.kubernetes.io/zone - podAntiAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - - labelSelector: - matchExpressions: - - key: type - operator: In - values: - - vmselect - - vminsert - - tools - topologyKey: kubernetes.io/hostname - namespaceSelector: {} - - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" - -service: - type: ClusterIP - port: 80 - -deployments: - - nodepool: vminsert - bu: datascience - cpuRequest: 1 - memoryRequest: 1Gi - replicaCount: 3 - - nodepool: vmselect - bu: datascience - cpuRequest: 1 - memoryRequest: 1Gi - replicaCount: 3 diff --git a/helm-overrides/k8s-datascience-prd-ase1/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index f0ab538..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,496 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-dataengg-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "datascience" - team: "datascience-sre" - service: "node-exporter-datascience-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-datascience-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index c33a4fc..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-datascience-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-datascience-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "datascience" - team: "datascience-sre" - service: "stackdriver-exporter-datascience-prd" - env: "prd" - priority: "p0" - type: "exporter" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-datascience-prd-0622" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,logging.googleapis.com/user,run.googleapis.com/,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime,compute.googleapis.com/instance' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - dedicated: vmselect - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-dssre-stackdriver-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/k8s-datascience-prd-ase1/telegraf-operator/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/telegraf-operator/custom-values.yaml deleted file mode 100644 index 2904468..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,150 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "datascience" - team: "datascience-sre" - service: "telegraf-operator-datascience-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - dedicated: "vmselect" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false diff --git a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml deleted file mode 100644 index 000187a..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml +++ /dev/null @@ -1,295 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-datascience-prd-dr -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-datascience-prd-dr-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent-dr - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - minReadySeconds: 30 - progressDeadlineSeconds: 60 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-dssre-vmagent-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - - https://vminsert-infra-prd-dr.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vmagent-datascience-prd" - env: "prd" - priority: "p0" - type: "vmagent-dr" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-datascience-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: contour-internal-1 - annotations: {} - # nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - # nginx.ingress.kubernetes.io/rewrite-target: / - # nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-datascience-prd-dr.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 10 - memory: 10Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent-dr" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml deleted file mode 100644 index 2839733..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml +++ /dev/null @@ -1,272 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 1 - -fullnameOverride: vmagent-datascience-prd-fb -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-datascience-prd-fb-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - -# Remove topology spread constraints -topologySpreadConstraints: [] - -# Modify deployment strategy -deployment: - enabled: true - -# Remove HPA -horizontalPodAutoscaler: - enabled: false - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-infr-sre-vmagent-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -# Remove pod disruption budget -podDisruptionBudget: - enabled: false - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-datascience.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vmsinglenode-infra-prd.victoriametrics.svc.clusterset.local:8428/api/v1/write - - https://vmsinglenode-infra-prd.meeshogcp.in/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "datascience-fb" - team: "sre" - service: "vmagent-datascience-prd-fb" - env: "prd" - priority: "p0" - type: "vmagent-fb" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-datascience-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-datascience-prd-fb.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -# Modify resources -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - requests: - cpu: 2 - memory: 5Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmselect" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-agent/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index 73cac50..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,295 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-datascience-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-datascience-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - minReadySeconds: 30 - progressDeadlineSeconds: 60 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-sre-vmagnt-prd-mds@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-datascience.meeshogcp.in/insert/multitenant/prometheus/api/v1/write - - http://vminsert-datascience-prd.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vmagent-datascience-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-datascience-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: contour-internal-1 - annotations: {} - # nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - # nginx.ingress.kubernetes.io/rewrite-target: / - # nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-datascience-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 40 - memory: 80Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent-mds" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-mds" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml deleted file mode 100644 index 3b01e27..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert-dr - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-datascience-prd-dr - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/datascience/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-datascience-prd-dr.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-datascience-prd-dr" - env: "prd" - priority: "p0" - type: "vmalert-dr" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - dedicated: "vmagent-dr" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert-secured-stateful/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert-secured-stateful/custom-values.yaml deleted file mode 100644 index 00b2d6d..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert-secured-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-stateful-secured-datascience-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-datascience-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-datascience-prd.victoriametrics.svc.cluster.local:8480/insert/200/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-datascience-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/sensitive/datascience/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: false - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-datascience-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "datascience" - team: "sre" - service: "vmalert-stateful-secured-datascience-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-secured-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert-secured/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert-secured/custom-values.yaml deleted file mode 100644 index 3866281..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert-secured/custom-values.yaml +++ /dev/null @@ -1,305 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert-secured - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-secured-datascience-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-datascience-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-datascience-prd.victoriametrics.svc.cluster.local:8480/insert/200/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-datascience-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config1/sensitive/datascience/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-secured-datascience-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-secured-datascience-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml deleted file mode 100644 index b34b19b..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-datascience-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-datascience-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-datascience-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-datascience-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/datascience/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-datascience-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "datascience" - team: "sre" - service: "vmalert-datascience-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert/custom-values.yaml deleted file mode 100644 index 5232063..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-alert/custom-values.yaml +++ /dev/null @@ -1,305 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-datascience-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-datascience-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-datascience-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-datascience-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config1/datascience/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-datascience-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-datascience-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-insert/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index 1322278..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,234 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-datascience-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "high-priority" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-datascience-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vminsert-datascience-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 7 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vminsert-mds" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 10 - memory: 10Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: true - # -- Ingress annotations - annotations: - # nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - # nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-datascience-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: contour-internal-0 - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-select/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index f84d4b7..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,293 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-datascience-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "high-priority" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-datascience-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 150 - clusternative.maxConcurrentRequests: 256 - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vmselect-datascience-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 15 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "vmselect" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 20 - memory: 31Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: contour-internal-0 - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-datascience.prd.meesho.int - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-storage/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index 38a887b..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,316 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-datascience-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage-sale-24aug" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage-sale-24aug" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 250Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vmstorage-datascience-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 24 - memory: 330Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.93.7-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-datascience-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-datascience-prd-ase1/victoriametrics-agent/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index 1ac1cb7..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,485 +0,0 @@ -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "datascience-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -# -- Replica count -replicaCount: 2 - -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - -# -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables -deployment: - spec: - minReadySeconds: 1200 - progressDeadlineSeconds: 1800 - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - -image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - # -- Image tag, set to `Chart.AppVersion` by default - tag: v1.133.0 # rewrites Chart.AppVersion - # -- Variant of the image to use. - # e.g. enterprise, scratch - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - -# -- Image pull secrets -imagePullSecrets: [] - -# -- Add additional DNS entries to pods hosts file. Check [official documentation](https://kubernetes.io/docs/tasks/network/customize-hosts-file-for-pods/) -hostAliases: [] -# - ip: 192.168.1.1 -# hostNames: -# - test.example.com -# - another.example.net - -# -- Override chart name -nameOverride: "" - -# -- Override resources fullname -fullnameOverride: "vm-agent-datascience-prd" - -# -- Container working directory -containerWorkingDir: "/" - -rbac: - # -- Enables Role/RoleBinding creation - create: true - - # -- Role/RoleBinding annotations - annotations: {} - - # -- Role/RoleBinding labels - extraLabels: {} - - # -- If true and `rbac.enabled`, will deploy a Role/RoleBinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - - # -- additional rules for a role - extraRules: [] - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - # -- Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-sre-vmagnt-prd-mds@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # -- mount API token to pod directly - automountToken: true - -# -- See `kubectl explain poddisruptionbudget.spec` for more or check [official documentation](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# -- Generates `remoteWrite.*` flags and config maps with value content for values, that are of type list of map. -# Each item should contain `url` param to pass validation. -remoteWrite: - - url: http://vm-insert-datascience-prd.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write - disableOnDiskQueue: false - dropSamplesOnOverload: false - - url: http://prd-census-server-datascience.prd-census-server-datascience.svc.cluster.local/api/v1/write - disableOnDiskQueue: true - dropSamplesOnOverload: true -# urlRelabelConfig: -# - action: keep -# source_labels: [env] -# regex: "dev" - -# -- VMAgent extra command line arguments -extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8429 - promscrape.config.strictParse: false # default is true - promscrape.maxScrapeSize: 1000000000 # default is 167772160 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - # promscrape.suppressScrapeErrors: true # default is false - - # Uncomment and specify the port if you want to support any of the protocols: - # https://docs.victoriametrics.com/victoriametrics/vmagent/#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for more details. -env: - - name: GOGC - value: "200" - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth-secret - # key: password - -# -- Specify alternative source for env variables -envFrom: - [] - #- configMapRef: - # name: special-config - -# -- Extra labels for Deployment and Statefulset -extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-agent-datascience-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra labels for Pods only -podLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-agent-datascience-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra selector labels common for pod and service -selectorLabels: {} - -# -- Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# -- Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# -- Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -# -- Extra containers to run in a pod with vmagent -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -# -- Init containers for vmagent -initContainers: - [] - # - name: example - # image: example-image - -# -- Security context to be added to pod -podSecurityContext: - enabled: true - # fsGroup: 2000 - -# -- Security context to be added to pod's containers -securityContext: - enabled: true - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - # -- Enable agent service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - extraLabels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) for details - externalIPs: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8429 - # -- Target port - targetPort: http - # nodePort: 30000 - # -- Service type - type: ClusterIP - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - -ingress: - # -- Enable deployment of ingress for agent - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-agent-datascience-prd.meeshogcp.in - path: - - / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - - # -- Ingress controller class name - ingressClassName: "nginx-internal" - - # -- Ingress path type - pathType: Prefix - -# -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) -resources: - requests: - cpu: 70 - memory: 135Gi - -# -- Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) -nodeSelector: - dedicated: "vmagent-n4d" - -# -- Node tolerations for server scheduling to nodes with taints. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-n4d" - effect: "NoSchedule" - -# -- Pod topologySpreadConstraints -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# -- Pod affinity -affinity: {} - -# -- VMAgent [scraping configuration](https://docs.victoriametrics.com/victoriametrics/vmagent/#how-to-collect-metrics-in-prometheus-format) -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-datascience-prd" - -# -- Priority class to be assigned to the pod(s) -priorityClassName: "" - -# -- Enable the host network -hostNetwork: false - -serviceMonitor: - # -- Enable deployment of Service Monitor for server component. This is Prometheus operator object - enabled: false - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Service Monitor relabelings - relabelings: [] - # -- Basic auth params for Service Monitor - basicAuth: {} - # -- Service Monitor metricRelabelings - metricRelabelings: [] - # -- Service Monitor targetPort - targetPort: http - # interval: 15s - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - -# -- Empty dir configuration for a case, when persistence is disabled -emptyDir: {} - -persistentVolume: - # -- Create/use Persistent Volume Claim for server component. Empty dir if false - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Size of the volume. Should be calculated based on the logs you send and retention policy you set. - size: 10Gi - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume additional labels - extraLabels: {} - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Horizontal Pod Autoscaling. -# Note that it is not intended to be used for vmagents which perform scraping. -# In order to scale scraping vmagents check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) -horizontalPodAutoscaling: - # -- Use HPA for vmagent - enabled: false - # -- Maximum replicas for HPA to use to to scale vmagent - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale vmagent - minReplicas: 1 - # -- Metric for HPA to use to scale vmagent - metrics: [] - -# -- VMAgent scrape configuration -config: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -probe: - # -- Readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - # -- Liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - # -- Startup probe - startup: {} - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -allowedMetricsEndpoints: - - /metrics - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-datascience-prd-ase1/victoriametrics-insert/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/victoriametrics-insert/custom-values.yaml deleted file mode 100644 index abc0f30..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/victoriametrics-insert/custom-values.yaml +++ /dev/null @@ -1,411 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: false - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- IDs of vmstorage nodes to exclude from writing - excludeStorageIDs: [] - # -- Override default `app` label name - name: vminsert - # -- VMInsert strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - # override Chart.AppVersion - tag: v1.107.0-cluster - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMInsert http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-datascience-prd - # -- Extra command line arguments for vminsert component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8480 - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-datascience-prd-0.vm-storage-datascience-prd.victoriametrics.svc:8400" - - "vm-storage-datascience-prd-1.vm-storage-datascience-prd.victoriametrics.svc:8400" - - "vm-storage-datascience-prd-2.vm-storage-datascience-prd.victoriametrics.svc:8400" - - "vm-storage-datascience-prd-3.vm-storage-datascience-prd.victoriametrics.svc:8400" - - "vm-storage-datascience-prd-4.vm-storage-datascience-prd.victoriametrics.svc:8400" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-insert-datascience-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Pod's additional labels - podLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-insert-datascience-prd" - env: "prd" - priority: "p0" - type: "vminsert" - terminationGracePeriodSeconds: 30 - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - - # -- Readiness & Liveness probes - probe: - # -- VMInsert readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert startup probe - startup: {} - - # -- Relabel configuration - relabel: - enabled: false - config: [] - # -- Use existing configmap if specified - # otherwise .config values will be used. Relabel config **should** reside under `relabel.yml` key - configMap: "" - - # Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 7 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vminsert - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vminsert - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - dedicated: "vminsert-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 14 - memory: 10Gi - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - service: - # -- Create VMInsert service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here]( https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) for details - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-insert-datascience-prd.meeshogcp.in - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - - # -- Ingress controller class name - ingressClassName: nginx-internal - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for insert component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/insert" .Values.vminsert }}' - - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vminsert component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-datascience-prd-ase1/victoriametrics-select/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/victoriametrics-select/custom-values.yaml deleted file mode 100644 index 7f2ad5d..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/victoriametrics-select/custom-values.yaml +++ /dev/null @@ -1,445 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - splitService: true - externalService: - enabled: true - name: vmselect-datascience-prd-proxy - # -- Override default `app` label name - clusternativeService: - enabled: false - name: "" - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # override Chart.AppVersion - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMSelect http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-datascience-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - # -- Extra command line arguments for vmselect component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8481 - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-datascience-prd-0.vm-storage-datascience-prd.victoriametrics.svc:8401" - - "vm-storage-datascience-prd-1.vm-storage-datascience-prd.victoriametrics.svc:8401" - - "vm-storage-datascience-prd-2.vm-storage-datascience-prd.victoriametrics.svc:8401" - - "vm-storage-datascience-prd-3.vm-storage-datascience-prd.victoriametrics.svc:8401" - - "vm-storage-datascience-prd-4.vm-storage-datascience-prd.victoriametrics.svc:8401" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-select-datascience-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # -- Pod's additional labels - podLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-select-datascience-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Readiness & Liveness probes - probe: - # -- VMSelect readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 40 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 10 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vmselect - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vmselect - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - dedicated: "vmselect" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 38 - memory: 70Gi - - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Create VMSelect service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: false - - # -- Ingress annotations - annotations: {} - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-select-datascience.prd.meesho.int - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # -- Ingress controller class name - ingressClassName: contour-internal-0 - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for select component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/select" .Values.vmselect }}' - - # -- vmselect mode: deployment, daemonSet - mode: deployment - - # -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables - deployment: - spec: - # -- VMSelect strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - - # -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables - statefulSet: - enabled: false - spec: - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume extra labels - extraLabels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmselect component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: false - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - diff --git a/helm-overrides/k8s-datascience-prd-ase1/victoriametrics-storage/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1/victoriametrics-storage/custom-values.yaml deleted file mode 100644 index 5ea62bf..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1/victoriametrics-storage/custom-values.yaml +++ /dev/null @@ -1,363 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - enabled: false - -vminsert: - enabled: false - -vmauth: - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- Override default `app` label name - name: vmstorage - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Variant of the image to use. e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMStorage http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-datascience-prd - - # -- Additional environment variables (ex.: secret tokens, flags). Check https://docs.victoriametrics.com/victoriametrics/#environment-variables for details - env: [] - # -- Specify alternative source for env variables - envFrom: [] - # -- Data retention period. Possible units character: h(ours), d(ays), w(eeks), y(ears), if no unit character specified - month. The minimum retention period is 24h. - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - httpListenAddr: :8482 - loggerTimezone: "Asia/Kolkata" - dedup.minScrapeInterval: 60s - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage-n4d" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage-n4d" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - # -- Use an alternate scheduler, e.g. "stork". Check https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ for details - # schedulerName: - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Override Persistent Volume Claim name - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are https://kubernetes.io/docs/concepts/storage/persistent-volumes/ - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume extra labels - extraLabels: {} - # -- Storage class name. Will be empty if not set - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume - size: 8134Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-storage-datascience-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - # -- Pod's additional labels - podLabels: - bu: "datascience" - team: "datascience-sre" - service: "vm-storage-datascience-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Details are https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 42 - memory: 330Gi - - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: - enabled: false - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - podSecurityContext: - enabled: false - service: - enabled: true - # -- Service annotations - annotations: {} - # -- Service ClusterIP - clusterIP: None - # -- Service type - type: ClusterIP - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmselect - vmselectPort: 8401 - # -- Extra service ports - extraPorts: [] - # -- Health check node port for a service. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilyPolicy: "" - # -- List of service IP families. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilies: [] - # -- Traffic Distribution. Check https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution - trafficDistribution: "" - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - minReadySeconds: 5 - # -- Readiness probes - probe: - # -- VMStorage readiness probe - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMStorage startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmstorage component - enabled: false - # -- Maximum replicas for HPA to use to to scale the vmstorage component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmstorage component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmstorage component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: - scaleDown: - selectPolicy: Disabled - - vmbackupmanager: - # -- Enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-datascience-prd-ase1c/alloy/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/alloy/custom-values.yaml deleted file mode 100644 index f110f24..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/alloy/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -fullnameOverride: "alloy-datascience-prd" - -alloy: - configMap: - configFile: datascience.alloy - - clustering: - enabled: true - - extraPorts: - - name: "otlp-grpc" - port: 4317 - targetPort: 4317 - protocol: "TCP" - - name: "otlp-http" - port: 4318 - targetPort: 4318 - protocol: "TCP" - - resources: - requests: - cpu: 7 - memory: 55Gi - -configReloader: - enabled: true - -serviceAccount: - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-sre-grafna-obs-stk-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - -controller: - # Must be one of 'daemonset', 'deployment', or 'statefulset'. - type: 'deployment' - - nodeSelector: - dedicated: "alloy" - tolerations: - - key: "dedicated" - operator: "Equal" - value: "alloy" - effect: "NoSchedule" - - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 50 - targetCPUUtilizationPercentage: 80 - targetMemoryUtilizationPercentage: 80 \ No newline at end of file diff --git a/helm-overrides/k8s-datascience-prd-ase1c/aurva-dataplane/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/aurva-dataplane/custom-values.yaml deleted file mode 100644 index fd300b0..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/aurva-dataplane/custom-values.yaml +++ /dev/null @@ -1,652 +0,0 @@ -## Aurva Data Plane -## Ref: https://github.com/aurva-io/aurva-charts.git - -postgresql: - enabled: true - fullnameOverride: "aurva-dataplane-database" - volumePermissions: - ## @param volumePermissions.enabled Enable init container that changes the owner and group of the persistent volume - ## - enabled: true - global: - storageClass: pd-standard-retain - postgresql: - auth: - postgresPassword: "aurva" - database: "controller" - # Add toleration to make sure where this postgres db pod should reside (Applicable for production workloads): For more detail ref: https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/ - primary: - extendedConfiguration: | - max_connections = 300 - - # #PLACEHOLDER## - # tolerations: - # - effect: NoSchedule - # key: dedicated - # operator: Equal - # value: datascience-devops - - # # -- Select nodes to deploy which matches the following labels - # nodeSelector: ##PLACEHOLDER## - # dedicated: datascience-devops - -# -- Provide a name in place of `aurva` -# namespaceOverride: aurva-dataplane - -########################################################## -# Global Configs -########################################################## -global: - aurva_controller: - enabled: true - aurva_fastdet: - enabled: false - aurva_pii_analyzer: - enabled: true - aurva_ocr: - enabled: true - aurva_collector: - enabled: true - - deploymentAnnotations: {} - - priorityClassName: "" - -########################################################## -# Aurva Controller -########################################################## -aurva_controller: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "datascience" - team: "datascience-devops" - service: "aurva-datascience-prd" - env: "prd" - priority: "p0" - type: "aurva_controller" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 10 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - image: - # Image of the app container - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-controller/aurva-controller - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - - type: secret - name: aurva-controller-secrets - # - type: configmap - # name: proxy-datasource-config - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 2 - requests: - memory: 1Gi - cpu: 1 - - aurvaFastdet: - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-fastdet/aurva-fastdet - tag: "v2.30.13" - pullPolicy: IfNotPresent - envFrom: [] - env: [] - resources: - limits: - cpu: 0.5 - memory: 512Mi - requests: - cpu: 0.5 - memory: 512Mi - - nodeSelector: ##PLACEHOLDER## - dedicated: datascience-devops - - # -- Taint tolerations for nodes - ##PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: datascience-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-controller-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - config: - #variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - COMMAND_URL: "command.aurva-prd.meeshogcp.in:80" - DEPLOYMENT_TYPE: "kubernetes" - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - FLUSHER_WORKER_POOL_SIZE: "1000" - FLUSHER_BATCH_SIZE: "30000" - UNIQUENESS_IDENTIFIER: "k8s-datascience-prd-ase1" #Recommendation: should be equal to cluster name - PROVIDER_ACCOUNT_ID: "meesho-datascience-prd-0622" # GCP PROJECT ID (not Number) - REGION: "asia-southeast1" #eg: asia-south1 - ENVIRONMENT: "prod" - OCR_ENABLED: "true" - MONITORING_ENABLED: "false" - HYBRID_ONLY_MODE: "false" - FORCE_TLS: "false" - FASTDET_FLAG : "true" - AADHAAR_ENHANCER: "0" - WORKSPACE_EVENT_TRACKING_ENABLED: "false" - ACCESS_IQ_ENABLED: "true" - GRPC_ENFORCE_ALPN_ENABLED: "false" - ENABLE_FASTDET: "true" - GOGC: "70" - #pii data - PII_BUCKET_NAME: "gcs-infra-devop-aurva-datascience-prd" - PII_LOG_BUCKET_REGION: "asia-southeast1" - PII_LOG_CRON: "*/5 * * * *" - ENABLE_PII_LOG: "true" - #constants - SKIP_NAMESPACES: "contour-internal-1-datascience-prd,contour-internal-0-datascience-prd,contour-external-datascience-prd" - CLOUD_PROVIDER: "gcp" - LOG_ENV: "production" - RDS_SCANNER_AVAILABILITY : "false" - REDSHIFT_SCANNER_AVAILABILITY : "false" - S3_SCANNER_AVAILABILITY : "false" - DYNAMO_SCANNER_AVAILABILITY: "false" - DOCDB_SCANNER_AVAILABILITY: "false" - OPENSEARCH_SCANNER_AVAILABILITY: "false" - CLOUDSQL_SCANNER_AVAILABILITY: "true" - BIGQUERY_SCANNER_AVAILABILITY: "true" - AWS_SNAPSHOT_SCANNER_AVAILABILITY: "false" - CLOUDSTORAGE_SCANNER_AVAILABILITY: "true" - KEYSPACES_SCANNER_AVAILABILITY: "false" - ALLOYDB_SCANNER_AVAILABILITY: "true" - BIGTABLE_SCANNER_AVAILABILITY: "true" - GCP_BACKUP_AVAILABILITY: "true" - EGRESS_MODE_ONLY: "false" - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - SCAN_UUID_ENABLED: "true" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-controller-sa - # -- Annotations applied to created service account - annotations: - iam.gke.io/gcp-service-account: sa-datascience-prd-aurva-contr@meesho-datascience-prd-0622.iam.gserviceaccount.com -# eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 15 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 - - -########################################################## -# Aurva OCR -########################################################## -aurva_ocr: - # -- Additional labels for aurva-controller - additionalLabels: - bu: "datascience" - team: "datascience-devops" - service: "aurva-datascience-prd" - env: "prd" - priority: "p0" - type: "aurva_ocr" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 1 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-ocr/aurva-ocr - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-ocr: - type: secret - name: aurva-ocr-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 1 - requests: - memory: 2Gi - cpu: 1 - - # -- Select nodes to deploy which matches the following labels - - nodeSelector: ##PLACEHOLDER## - dedicated: datascience-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: datascience-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-ocr-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - OCR_TIME_LIMIT: "1" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-ocr-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # iam.gke.io/gcp-service-account: service-account@gcp.iam.gserviceaccount.com - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva Collector -########################################################## -aurva_collector: - - # -- Additional labels for aurva-analyzer - additionalLabels: - bu: "datascience" - team: "datascience-devops" - service: "aurva-datascience-prd" - env: "prd" - priority: "p0" - type: "aurva_collector" - - - # -- Annotations on aurva-analyzer - annotations: {} -# "key": "value" - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-collector/aurva-collector - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-controller: - type: secret - name: aurva-collector-secrets - - resources: - limits: - cpu: 800m - memory: 800Mi - requests: - cpu: 200m - memory: 512Mi - - podSecurityContext: {} - - securityContext: - privileged: false - capabilities: - add: - # For kernel v5.8 and above we don't need CAP_SYS_ADMIN or CAP_SYS_RESOURCE - # we just need CAP_BPF and CAP_PERFMON. This has been tested on our EKS node - # which is on kernel v5.10.x - # When SSL Tracing is required we need CAP_SYS_ADMIN and CAP_SYS_PTRACE - # on top of the previous capabilities - # So finally these are the 4 possible combinations for capabilies - # 1. Newer Kernels without SSL - - BPF - - PERFMON - # 2. Newer Kernels with SSL - # - SYS_ADMIN - # - SYS_PTRACE - # 3. Older Kernels without SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # 4. Older Kernels with SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # - SYS_PTRACE - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - volumes: - - name: debugfs - mountPath: /sys/kernel/debug - hostPath: /sys/kernel/debug - - name: vmlinux - mountPath: /sys/kernel/btf/vmlinux - hostPath: /sys/kernel/btf/vmlinux - - name: procfs - mountPath: /host/proc - hostPath: /proc - - # -- Taint tolerations for nodes - tolerations: - # - effect: NoSchedule - # key: dedicated - # operator: Equal - # value: megatetra - - operator: Exists - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external - - alloy - - preprod-spot-16 - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-collector-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - # variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - TRACE_INTERNAL_SVC: "true" - TRACE_INTERNAL_SVC_HTTP: "true" - INTERNAL_SVC_SAMPLE_INTERVAL: "2m" - TRACE_SSL: "false" - TRACE_PSQL: "false" - TRACE_SQLSERVER: "false" - TRACE_MYSQL: "false" - TRACE_EGRESS: "true" - TRACE_GO_TLS: "false" - TRACE_ML_SERVICES: "false" - # constants - LOG_ENV: production - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - MONITORING_ENABLED: "false" - ENABLE_INGRESS_INFORMER: "false" - ENABLE_SERVICE_INFORMER: "false" - ENABLE_ISTIO_INFORMER: "false" - EXCLUDED_PII_REGEX_TYPES: "ip_address,us_bank_number,us_driver_license,us_itin,us_passport,us_routing,us_mbi,ssn" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-collector-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva PII Analyzer -########################################################## -aurva_pii_analyzer: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "datascience" - team: "datascience-devops" - service: "aurva-datascience-prd" - env: "prd" - priority: "p0" - type: "aurva_pii_analyzer" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 3 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - ##PLACEHOLDER## - nodeSelector: - dedicated: datascience-devops - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-piianalyzer/aurva-piianalyzer - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-pii-analyzer: - type: secret - name: aurva-pii-analyzer-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 4Gi - cpu: 4 - requests: - memory: 2Gi - cpu: 2 - - nodeSelector: ##PLACEHOLDER## - dedicated: datascience-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: datascience-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-pii-analyzer-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - SCHEDULER_TIME: "1" - SUPPORTED_REGION: "US" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-pii-analyzer-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 3 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 diff --git a/helm-overrides/k8s-datascience-prd-ase1c/conntrack-adjuster/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 7b64b15..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - # Maximum number of conntrack entries - max: 2097152 - # Hash size for conntrack - hashsize: 524288 - # Sleep interval between adjustments (seconds) - sleepInterval: 30 - -# Enable additional matchExpressions -# addExtraMatchExpressions: true - -# Additional matchExpressions to append -# additionalMatchExpressions: -# - key: node-role -# operator: In -# values: -# - "worker" -# - "ingress" -# - key: environment -# operator: In -# values: -# - "production" -# - "staging" \ No newline at end of file diff --git a/helm-overrides/k8s-datascience-prd-ase1c/contour-cert-checker/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/contour-cert-checker/custom-values.yaml deleted file mode 100644 index f6ef9ef..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=k8s-datascience-prd-ase1c"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - dedicated: datascience-devops - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: datascience-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/k8s-datascience-prd-ase1c/contour-internal-0/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/contour-internal-0/custom-values.yaml deleted file mode 100644 index 8bdb749..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,94 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 60s - max-connection-duration: 600s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 500m - memory: 3Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 5% - maxUnavailable: 0 - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 8Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-dsci-c-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-datascience-prd-ase1c/contour-internal-1/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/contour-internal-1/custom-values.yaml deleted file mode 100644 index 53978fa..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,110 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 400s - max-connection-duration: 540s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 8 - memory: 10Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - logLevel: error - terminationGracePeriodSeconds: 500 - autoscaling: - behaviour: - scaledown: - policies: - - periodseconds: 500 - type: Pods - value: 1 - selectpolicy: Min - stabilizationWindowSeconds: 600 - scaleup: - policies: - - periodseconds: 15 - type: Pods - value: 2 - selectpolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "35" - targetMemory: "40" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 18 - memory: 12Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int1-dsci-c-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-datascience-prd-ase1c/contour-internal-intra-0/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index 8d2eff2..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,92 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 60s - max-connection-duration: 600s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 500m - memory: 3Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 5% - maxUnavailable: 0 - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 8Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-datascience-prd-ase1c/contour-internal-intra-1/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index 527e05d..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,108 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 400s - max-connection-duration: 540s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: datascience - team: datascience-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 8 - memory: 10Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: datascience - team: datascience-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - logLevel: error - terminationGracePeriodSeconds: 500 - autoscaling: - behaviour: - scaledown: - policies: - - periodseconds: 500 - type: Pods - value: 1 - selectpolicy: Min - stabilizationWindowSeconds: 600 - scaleup: - policies: - - periodseconds: 15 - type: Pods - value: 2 - selectpolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "35" - targetMemory: "40" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 18 - memory: 12Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-datascience-prd-ase1c/coredns/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/coredns/custom-values.yaml deleted file mode 100644 index f54cbf6..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/coredns/custom-values.yaml +++ /dev/null @@ -1,30 +0,0 @@ -replicaCount: 16 - -labels: - bu: datascience - team: datascience-devops - env: prd - -clusterIP: 10.217.16.2 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - -nodeSelector: - dedicated: datascience-devops - kubernetes.io/os: linux - -overwriteRegion: "c" -communicationType: "" \ No newline at end of file diff --git a/helm-overrides/k8s-datascience-prd-ase1c/external-secrets/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/external-secrets/custom-values.yaml deleted file mode 100644 index b7dd462..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/external-secrets/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Additional service annotations - annotations: {} - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: datascience-devops - nodeSelector: - dedicated: datascience-devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: datascience-devops - nodeSelector: - dedicated: datascience-devops - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: datascience-devops - nodeSelector: - dedicated: datascience-devops diff --git a/helm-overrides/k8s-datascience-prd-ase1c/flagger/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/flagger/custom-values.yaml deleted file mode 100644 index b446c1f..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/flagger/custom-values.yaml +++ /dev/null @@ -1,63 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - dedicated: datascience-devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: datascience-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: datascience-devops - bu: infra - region: ase1c - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-datascience-rollout-service.prd-datascience-rollout-service.svc.cluster.local" - -rolloutService: - enabled: true - apps: diff --git a/helm-overrides/k8s-datascience-prd-ase1c/fluentd/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/fluentd/custom-values.yaml deleted file mode 100644 index e0c25a0..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/fluentd/custom-values.yaml +++ /dev/null @@ -1,718 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-dssre-fluentd-prd@meesho-dscience-ase1c-prd-0625.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: datascience - team: sre - type: fluentd - service: fluentd-datascience-prd - priority: p0 - env: prd - zone_extended: ase1c - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: datascience - team: sre - type: fluentd - service: fluentd-datascience-prd - priority: p0 - env: prd - zone_extended: ase1c - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: [] -# - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - # - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-datascience-prd-ase1c/ingress-nginx-internal/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/ingress-nginx-internal/custom-values.yaml deleted file mode 100644 index 589576a..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/ingress-nginx-internal/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 6 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dsci-int-ase1c-prd"}}}' - nodeSelector: - dedicated: nginx-internal - tolerations: - - key: "dedicated" - operator: "Equal" - value: "nginx-internal" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-datascience-prd-ase1c/keda/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/keda/custom-values.yaml deleted file mode 100644 index d4cee29..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/keda/custom-values.yaml +++ /dev/null @@ -1,46 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: datascience-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - podLabels: - bu: "datascience" - team: "datascience-devops" - metricsAdapter: - bu: "datascience" - team: "datascience-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 300m - memory: 450Mi - # -- Manage [resource request & limits] of KEDA metrics apiserver pod - metricServer: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 100m - memory: 200Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 100m - memory: 180Mi - requests: - cpu: 60m - memory: 120Mi - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' \ No newline at end of file diff --git a/helm-overrides/k8s-datascience-prd-ase1c/kube-dns/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/kube-dns/custom-values.yaml deleted file mode 100644 index 20b06d1..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.217.16.2"],"prd.mrouter.int.svc.cluster.local":["10.217.16.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.217.16.2"]} diff --git a/helm-overrides/k8s-datascience-prd-ase1c/kube-events/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/kube-events/custom-values.yaml deleted file mode 100644 index 4db407b..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-datascience-ase1c-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - dedicated: datascience-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: datascience-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: datascience-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: datascience-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: datascience-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: datascience-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-datascience-prd-ase1c/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 0c34b33..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-datascience-ase1c-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-datascience-ase1c-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "datascience" - team: "datascience-sre" - service: "kube-state-metrics-datascience-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "vmselect" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 17m - memory: 1.5Gi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-datascience-prd-ase1c/kubernetes-dashboard/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/kubernetes-dashboard/custom-values.yaml deleted file mode 100644 index cd5b892..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/kubernetes-dashboard/custom-values.yaml +++ /dev/null @@ -1,442 +0,0 @@ -# Copyright 2017 The Kubernetes Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -# General configuration shared across resources -app: - # Mode determines if chart should deploy a full Dashboard with all containers or just the API. - # - dashboard - deploys all the containers - # - api - deploys just the API - mode: 'dashboard' - image: - pullPolicy: IfNotPresent - pullSecrets: [] - scheduling: - # Node labels for pod assignment - # Ref: https://kubernetes.io/docs/user-guide/node-selection/ - nodeSelector: {} - security: - # Allow overriding csrfKey used by API/Auth containers. - # It has to be base64 encoded random 256 bytes string. - # If empty, it will be autogenerated. - csrfKey: ~ - # SecurityContext to be added to pods - # To disable set the following configuration to null: - # securityContext: null - securityContext: - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - # ContainerSecurityContext to be added to containers - # To disable set the following configuration to null: - # containerSecurityContext: null - containerSecurityContext: - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - runAsUser: 1001 - runAsGroup: 2001 - capabilities: - drop: ["ALL"] - # Pod Disruption Budget configuration - # Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - minAvailable: 0 - maxUnavailable: 0 - networkPolicy: - enabled: false - ingressDenyAll: false - # Raw network policy spec that overrides predefined spec - # Example: - # spec: - # egress: - # - ports: - # - port: 123 - spec: {} - - # Common labels & annotations shared across all deployed resources - labels: {} - annotations: {} - # Common priority class used for all deployed resources - priorityClassName: null - settings: - ## Global dashboard settings - global: - # # Cluster name that appears in the browser window title if it is set - # clusterName: "" - # # Max number of items that can be displayed on each list page - # itemsPerPage: 10 - # # Max number of labels that are displayed by default on most views. - # labelsLimit: 3 - # # Number of seconds between every auto-refresh of logs - # logsAutoRefreshTimeInterval: 5 - # # Number of seconds between every auto-refresh of every resource. Set 0 to disable - # resourceAutoRefreshTimeInterval: 10 - # # Hide all access denied warnings in the notification panel - # disableAccessDeniedNotifications: false - # # Hide all namespaces option in namespace selection dropdown to avoid accidental selection in large clusters thus preventing OOM errors - # hideAllNamespaces: false - # # Namespace that should be selected by default after logging in. - # defaultNamespace: default - # # List of namespaces that should be presented to user without namespace list privileges. - # namespaceFallbackList: - # - default - ## Pinned resources that will be displayed in dashboard's menu - pinnedResources: [] - # - kind: customresourcedefinition - # # Fully qualified name of a CRD - # name: prometheus.monitoring.coreos.com - # # Display name - # displayName: Prometheus - # # Is this CRD namespaced? - # namespaced: true - ingress: - enabled: false - hosts: - # Keep 'localhost' host only if you want to access Dashboard using 'kubectl port-forward ...' on: - # https://localhost:8443 - - localhost - # - kubernetes.dashboard.domain.com - ingressClassName: internal-nginx - # Use only if your ingress controllers support default ingress classes. - # If set to true ingressClassName will be ignored and not added to the Ingress resources. - # It should fall back to using IngressClass marked as the default. - useDefaultIngressClass: false - # This will append our Ingress with annotations required by our default configuration. - # nginx.ingress.kubernetes.io/backend-protocol: "HTTPS" - # nginx.ingress.kubernetes.io/ssl-passthrough: "true" - # nginx.ingress.kubernetes.io/ssl-redirect: "true" - useDefaultAnnotations: true - pathType: ImplementationSpecific - # If path is not the default (/), rewrite-target annotation will be added to the Ingress. - # It allows serving Kubernetes Dashboard on a sub-path. Make sure that the configured path - # does not conflict with gateway route configuration. - path: / - issuer: - name: selfsigned - # Scope determines what kind of issuer annotation will be used on ingress resource - # - default - adds 'cert-manager.io/issuer' - # - cluster - adds 'cert-manager.io/cluster-issuer' - # - disabled - disables cert-manager annotations - scope: default - tls: - enabled: true - # If provided it will override autogenerated secret name - secretName: "" - labels: {} - annotations: {} - # Use the following toleration if Dashboard can be deployed on a tainted control-plane nodes - # - key: node-role.kubernetes.io/control-plane - # effect: NoSchedule - tolerations: [] - affinity: {} - -auth: - role: auth - image: - repository: docker.io/kubernetesui/dashboard-auth - tag: 1.2.2 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: auth - containerPort: 8000 - protocol: TCP - args: [] - env: [] - volumeMounts: - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for Auth related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -# API deployment configuration -api: - role: api - image: - repository: docker.io/kubernetesui/dashboard-api - tag: 1.10.1 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: api - containerPort: 8000 - protocol: TCP - # Additional container arguments - # Full list of arguments: https://github.com/kubernetes/dashboard/blob/master/docs/common/arguments.md - # args: - # - --system-banner="Welcome to the Kubernetes Dashboard" - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store exec logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for API related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -# WEB UI deployment configuration -web: - role: web - image: - repository: docker.io/kubernetesui/dashboard-web - tag: 1.6.0 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: web - containerPort: 8000 - protocol: TCP - # Additional container arguments - # Full list of arguments: https://github.com/kubernetes/dashboard/blob/master/docs/common/arguments.md - # args: - # - --system-banner="Welcome to the Kubernetes Dashboard" - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for WEB UI related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -### Metrics Scraper -### Container to scrape, store, and retrieve a window of time from the Metrics Server. -### refs: https://github.com/kubernetes/dashboard/tree/master/modules/metrics-scraper -metricsScraper: - enabled: true - role: metrics-scraper - image: - repository: docker.io/kubernetesui/dashboard-metrics-scraper - tag: 1.2.1 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - containerPort: 8000 - protocol: TCP - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - livenessProbe: - httpGet: - scheme: HTTP - path: / - port: 8000 - initialDelaySeconds: 30 - timeoutSeconds: 30 - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for Metrics Scraper related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -## Optional Metrics Server sub-chart configuration -## Enable this if you don't already have metrics-server enabled on your cluster and -## want to use it with dashboard metrics-scraper -## refs: -## - https://github.com/kubernetes-sigs/metrics-server -## - https://github.com/kubernetes-sigs/metrics-server/tree/master/charts/metrics-server -metrics-server: - enabled: false - args: - - --kubelet-preferred-address-types=InternalIP - - --kubelet-insecure-tls - -## Required Kong sub-chart with DBless configuration to act as a gateway -## for our all containers. -kong: - enabled: false - admin: - tls: - enabled: false - ## Configuration reference: https://docs.konghq.com/gateway/3.6.x/reference/configuration - env: - dns_order: LAST,A,CNAME,AAAA,SRV - plugins: 'off' - nginx_worker_processes: 1 - ingressController: - enabled: false - manager: - enabled: false - dblessConfig: - configMap: kong-dbless-config - proxy: - type: ClusterIP - http: - enabled: true - -## Optional Cert Manager sub-chart configuration -## Enable this if you don't already have cert-manager enabled on your cluster. -cert-manager: - enabled: false - installCRDs: true - -## Optional Nginx Ingress sub-chart configuration -## Enable this if you don't already have nginx-ingress enabled on your cluster. -nginx: - enabled: false - controller: - electionID: ingress-controller-leader - ingressClassResource: - name: internal-nginx - default: false - controllerValue: k8s.io/internal-ingress-nginx - service: - type: ClusterIP - -## Extra configurations: -## - manifests -## - predefined roles -## - prometheus -## - etc... -extras: - # Extra Kubernetes manifests to be deployed - # manifests: - # - apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: additional-configmap - # data: - # mykey: myvalue - manifests: [] - serviceMonitor: - # Whether to create a Prometheus Operator service monitor. - enabled: false - # Here labels can be added to the serviceMonitor - labels: {} - # Here annotations can be added to the serviceMonitor - annotations: {} - # metrics.serviceMonitor.metricRelabelings Specify Metric Relabelings to add to the scrape endpoint - # ref: https://github.com/coreos/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # metrics.serviceMonitor.relabelings [array] Prometheus relabeling rules - relabelings: [] - # ServiceMonitor connection scheme. Defaults to HTTPS. - scheme: https - # ServiceMonitor connection tlsConfig. Defaults to {insecureSkipVerify:true}. - tlsConfig: - insecureSkipVerify: true diff --git a/helm-overrides/k8s-datascience-prd-ase1c/loadtester/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/loadtester/custom-values.yaml deleted file mode 100644 index 76550f1..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/loadtester/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -replicaCount: 1 - -Namespace: loadtester - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/dev/devops/flagger-loadtester - tag: 0.30.0 - pullPolicy: IfNotPresent - pullSecret: - -podLabels: - bu: datascience - team: datascience-shared - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - -podPriorityClassName: "" - -logLevel: info -cmd: - timeout: 1h - namespaceRegexp: "" - -nameOverride: "loadtester" -fullnameOverride: "" - -env: [] - -service: - type: ClusterIP - port: 80 - -resources: - requests: - cpu: 10m - memory: 64Mi - -volumes: [] -volumeMounts: [] - -nodeSelector: - dedicated: datascience-devops - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "datascience-devops" - effect: "NoSchedule" - -affinity: {} - -ingress: - slowStart: - enabled: false - enabled: true - ingressClassName: contour-internal-1 - servicePort: http - hosts: - - host: datascience-flagger-loadtester.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / -createContourGateway: true -contourResponseTimeout: false -rbac: - # rbac.create: `true` if rbac resources should be created - create: true - # rbac.scope: `cluster` to create cluster-scope rbac resources (ClusterRole/ClusterRoleBinding) - # otherwise, namespace-scope rbac resources will be created (Role/RoleBinding) - scope: - # rbac.rules: array of rules to apply to the role. example: - # rules: - # - apiGroups: [""] - # resources: ["pods"] - # verbs: ["list", "get"] - rules: [] - -# name of an existing service account to use - if not creating rbac resources -serviceAccountName: "" - -# App Mesh virtual node settings (to be used for AppMesh v1beta1) -meshName: "" -#backends: -# - app1.namespace -# - app2.namespace - -# App Mesh virtual node settings (to be used for AppMesh v1beta2) -appmesh: - enabled: false - backends: - - podinfo - - podinfo-canary - -#Istio virtual service and gatway settings. TLS secrets should be in namespace before enbaled it. ( secret format loadtester.fullname ) -istio: - enabled: false - host: flagger-loadtester.flagger - gateway: - enabled: false - tls: - enabled: false - httpsRedirect: false - -# when enabled, it will add a security context for the loadtester pod -securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 100 - runAsGroup: 101 - -podDisruptionBudget: - enabled: false - minAvailable: 1 diff --git a/helm-overrides/k8s-datascience-prd-ase1c/opentelemetry-coralogix/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/opentelemetry-coralogix/custom-values.yaml deleted file mode 100644 index 12794b1..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/opentelemetry-coralogix/custom-values.yaml +++ /dev/null @@ -1,281 +0,0 @@ -global: - domain: "coralogixsg.com" - defaultApplicationName: "default" - defaultSubsystemName: "nodes" - - # Old endpoint based configuration, - # please use domain instead. - traces: - endpoint: "" - metrics: - endpoint: "" - logs: - endpoint: "" - -# set distribution to openshift for openshift clusters -distribution: "" - -# Opentelemetry collector configuration -dedicatedValue: false -schedulerName: default-scheduler - -labels: - bu: "datascience" - team: "sre" - service: "opentelemetry-datascience-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: false - key: prd/admin/coralogix-keys - secretStoreRef: - name: vault-backend - -image: - # If you want to use the core image `otel/opentelemetry-collector`, you also need to change `command.name` value to `otelcol`. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - # repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - tag: "0.87.0" - # When digest is set to a non-empty value, images will be pulled by digest (regardless of tag value). - digest: "" -mode: daemonset -rollout: - rollingUpdate: - # When 'mode: daemonset', maxSurge cannot be used when hostPort is set for any of the ports - # maxSurge: 25% - maxUnavailable: 1 - strategy: RollingUpdate -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" -hostNetwork: true -dnsPolicy: "ClusterFirstWithHostNet" -fullnameOverride: opentelemetry-datascience-prd - -presets: - logsCollection: - enabled: false - storeCheckpoints: false - maxRecombineLogSize: 1048576 - extraFilelogOperators: [] -# - type: recombine -# combine_field: body -# source_identifier: attributes["log.file.path"] -# is_first_entry: body matches "^(YOUR-LOGS-REGEX)" - kubernetesAttributes: - enabled: false - hostMetrics: - enabled: false - kubeletMetrics: - enabled: false - -extraEnvs: -- name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" -- name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName -config: - extensions: {} - # zpages: - # endpoint: localhost:55679 - # pprof: - # endpoint: localhost:1777 - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 4s - tls: - insecure: true - sending_queue: - enabled: true - queue_size: 6000 - num_consumers: 200 - resolver: - dns: - hostname: opentelemetry-admin-prd.opentelemetry.svc.clusterset.local - # coralogix: - # timeout: "30s" - # private_key: "${CORALOGIX_PRIVATE_KEY}" - # domain: "{{.Values.global.domain}}" - # traces: - # endpoint: "{{ .Values.global.traces.endpoint }}" - # metrics: - # endpoint: "{{ .Values.global.metrics.endpoint }}" - # logs: - # endpoint: "{{ .Values.global.logs.endpoint }}" - # application_name_attributes: - # - "k8s.namespace.name" - # - "service.namespace" - # subsystem_name_attributes: - # - "k8s.deployment.name" - # - "k8s.statefulset.name" - # - "k8s.daemonset.name" - # - "k8s.cronjob.name" - # - "k8s.job.name" - # - "k8s.container.name" - # - "k8s.node.name" - # - "service.name" - # application_name: "{{.Values.global.defaultApplicationName }}" - # subsystem_name: "{{.Values.global.defaultSubsystemName }}" - processors: - k8sattributes: - filter: - node_from_env_var: KUBE_NODE_NAME - extract: - metadata: - - "k8s.namespace.name" - - "k8s.deployment.name" - - "k8s.statefulset.name" - - "k8s.daemonset.name" - - "k8s.cronjob.name" - - "k8s.job.name" - - "k8s.pod.name" - - "k8s.node.name" - memory_limiter: null # Will get the k8s resource limits - resourcedetection/env: - detectors: ["system","env"] - timeout: 2s - override: false - # spanmetrics: - # metrics_exporter: coralogix - # dimensions: - # - name: "k8s.deployment.name" - # - name: "k8s.statefulset.name" - # - name: "k8s.daemonset.name" - # - name: "k8s.cronjob.name" - # - name: "k8s.job.name" - # - name: "k8s.container.name" - # - name: "k8s.node.name" - # - name: "k8s.namespace.name" - receivers: - otlp: - protocols: - grpc: - endpoint: ${MY_POD_IP}:4317 - http: - endpoint: ${MY_POD_IP}:4318 - zipkin: - endpoint: ${MY_POD_IP}:9411 - jaeger: - protocols: - grpc: - endpoint: ${MY_POD_IP}:14250 - thrift_http: - endpoint: ${MY_POD_IP}:14268 - thrift_compact: - endpoint: ${MY_POD_IP}:6831 - thrift_binary: - endpoint: ${MY_POD_IP}:6832 - prometheus: - config: - scrape_configs: - - job_name: opentelemetry-collector - scrape_interval: 30s - static_configs: - - targets: - - ${MY_POD_IP}:8888 - service: - extensions: - # - zpages - # - pprof - - health_check -# - memory_ballast - telemetry: - metrics: - address: ${env:MY_POD_IP}:8888 - pipelines: - traces: - exporters: - - loadbalancing - processors: - - memory_limiter - # - spanmetrics - - batch - receivers: - - otlp - metrics: null - logs: null -tolerations: - - operator: Exists - -resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 1 - memory: 2G - -ports: - jaeger-binary: - enabled: true - containerPort: 6832 - servicePort: 6832 - hostPort: 6832 - protocol: TCP - # In order to enable podMonitor, following part must be enabled in order to expose the required port: - # metrics: - # enabled: true - -# podMonitor: -# enabled: true - -# prometheusRule: -# enabled: true -# defaultRules: -# enabled: true - -# Annotations to be added to pod -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-datascience-contour-ext - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-datascience-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert \ No newline at end of file diff --git a/helm-overrides/k8s-datascience-prd-ase1c/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index 348d96b..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,135 +0,0 @@ -fullnameOverride: opentelemetry-datascience-ase1c-prd - -mode: daemonset - -# priorityClassName: "system-node-critical" -# Removing since dataengg pods are going into pending state -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-central-prd-ase1c-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-datascience-contour-ext - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-datascience-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-datascience-default-prd-ase1c - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external - -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 100m - memory: 124Mi - limits: - cpu: 200m - memory: 256Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "datascience" - team: "sre" - service: "opentelemetry-datascience-ase1c-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - zone_extended: "ase1c" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/k8s-datascience-prd-ase1c/paused-container/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/paused-container/custom-values.yaml deleted file mode 100644 index ffb4e28..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/paused-container/custom-values.yaml +++ /dev/null @@ -1,82 +0,0 @@ -# Default values for hello-world. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -priorityClass: - name: "low-priority" - -image: - repository: nginx - tag: "1.14.2" - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - -nameOverride: "" -fullnameOverride: "paused-container-datascience-prd" - -extraLabels: - team: "devops" - bu: "datascience" - env: "prd" - service: "paused-container-datascience-prd" - priority: "p3" - type: "tools" - -topologySpreadConstraints: [] - # - labelSelector: - # matchLabels: - # dedicated: vminsert - # maxSkew: 1 - # topologyKey: topology.kubernetes.io/hostname - # whenUnsatisfiable: DoNotSchedule - -affinity: - podAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - - labelSelector: - matchExpressions: - - key: type - operator: In - values: - - vmselect - - vminsert - - tools - topologyKey: topology.kubernetes.io/zone - podAntiAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - - labelSelector: - matchExpressions: - - key: type - operator: In - values: - - vmselect - - vminsert - - tools - topologyKey: kubernetes.io/hostname - namespaceSelector: {} - - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" - -service: - type: ClusterIP - port: 80 - -deployments: - - nodepool: vminsert - bu: datascience - cpuRequest: 1 - memoryRequest: 1Gi - replicaCount: 3 - - nodepool: vmselect - bu: datascience - cpuRequest: 1 - memoryRequest: 1Gi - replicaCount: 3 diff --git a/helm-overrides/k8s-datascience-prd-ase1c/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index e81b5da..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,497 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-dataengg-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "datascience" - team: "datascience-sre" - service: "node-exporter-datascience-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - zone_extended: "ase1c" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-datascience-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index a2794ce..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,171 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-datascience-ase1c-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-datascience-ase1c-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "datascience" - team: "datascience-sre" - service: "stackdriver-exporter-datascience-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - zone_extended: "ase1c" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-datascience-ase1c-prd-0622" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,logging.googleapis.com/user,run.googleapis.com/,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime,compute.googleapis.com/instance' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - dedicated: vmselect - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-dssre-stackdriver-prd@meesho-datascience-ase1c-prd-0622.iam.gserviceaccount.com - } - - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/k8s-datascience-prd-ase1c/telegraf-operator/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/telegraf-operator/custom-values.yaml deleted file mode 100644 index 09596bf..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,151 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "datascience" - team: "datascience-sre" - service: "telegraf-operator-datascience-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - zone_extended: "ase1c" - -nodeSelector: - dedicated: "vmselect" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false diff --git a/helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-agent/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index def6627..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,296 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-datascience-ase1c-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-datascience-ase1c-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - minReadySeconds: 30 - progressDeadlineSeconds: 60 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-common-np-dsci-prd@meesho-datascience-ase1c-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-ase1c-prd-datascience.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-datascience-ase1c-prd.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vmagent-datascience-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmagent" - zone_extended: "ase1c" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-datascience-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: false - ingressClassName: contour-internal-1 - annotations: {} - # nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - # nginx.ingress.kubernetes.io/rewrite-target: / - # nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-datascience-ase1c-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 21 - memory: 14Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-alert/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-alert/custom-values.yaml deleted file mode 100644 index 5547eaa..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-alert/custom-values.yaml +++ /dev/null @@ -1,305 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-datascience-ase1c-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-datascience-ase1c-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-datascience-ase1c-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-datascience-ase1c-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/datascience/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-datascience-ase1c-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 250m - memory: 500Mi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-datascience-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmalert" - zone_extended: "ase1c" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - dedicated: "vmselect" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-insert/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index 00ddd1e..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,235 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-datascience-ase1c-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-datascience-ase1c-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vminsert-datascience-ase1c-prd" - env: "prd" - priority: "p0" - type: "vminsert" - zone_extended: "ase1c" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 20 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 1 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vminsert" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 3 - memory: 2Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - # nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - # nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-datascience-ase1c-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: contour-internal-0 - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-select/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index 89bf71d..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,299 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-datascience-ase1c-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-datascience-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 150 - clusternative.maxConcurrentRequests: 256 - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vmselect-datascience-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmselect" - zone_extended: "ase1c" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 1 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "vmselect" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 1 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 10 - memory: 16Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-datascience-ase1c-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - clusternativeService: - enabled: true - port: 8401 - sessionAffinity: None - ipFamilyPolicy: SingleStack - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-storage/custom-values.yaml b/helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index 5718ff8..0000000 --- a/helm-overrides/k8s-datascience-prd-ase1c/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,317 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-datascience-ase1c-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 250Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "datascience" - team: "datascience-sre" - service: "vmstorage-datascience-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - zone_extended: "ase1c" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 10 - memory: 105Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.93.7-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-datascience-ase1c-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-demand-prd-ase1/README.md b/helm-overrides/k8s-demand-prd-ase1/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/k8s-demand-prd-ase1/alloy/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/alloy/custom-values.yaml deleted file mode 100644 index da018f1..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/alloy/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -fullnameOverride: "alloy-demand-prd" - -alloy: - configMap: - configFile: demand.alloy - - clustering: - enabled: true - - extraPorts: - - name: "otlp-grpc" - port: 4317 - targetPort: 4317 - protocol: "TCP" - - name: "otlp-http" - port: 4318 - targetPort: 4318 - protocol: "TCP" - - resources: - requests: - cpu: 14 - memory: 115Gi - -configReloader: - enabled: true - -serviceAccount: - annotations: { - iam.gke.io/gcp-service-account: sa-dmnd-sre-grafna-obs-stk-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - } - -controller: - # Must be one of 'daemonset', 'deployment', or 'statefulset'. - type: 'deployment' - - nodeSelector: - dedicated: "alloy" - tolerations: - - key: "dedicated" - operator: "Equal" - value: "alloy" - effect: "NoSchedule" - - autoscaling: - enabled: true - minReplicas: 1 - maxReplicas: 50 - targetCPUUtilizationPercentage: 80 - targetMemoryUtilizationPercentage: 80 \ No newline at end of file diff --git a/helm-overrides/k8s-demand-prd-ase1/aurva-dataplane/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/aurva-dataplane/custom-values.yaml deleted file mode 100644 index 7030232..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/aurva-dataplane/custom-values.yaml +++ /dev/null @@ -1,675 +0,0 @@ -## Aurva Data Plane -## Ref: https://github.com/aurva-io/aurva-charts.git - -postgresql: - enabled: true - fullnameOverride: "aurva-dataplane-database" - volumePermissions: - ## @param volumePermissions.enabled Enable init container that changes the owner and group of the persistent volume - ## - enabled: true - global: - storageClass: pd-standard-retain - postgresql: - auth: - postgresPassword: "aurva" - database: "controller" - # Add toleration to make sure where this postgres db pod should reside (Applicable for production workloads): For more detail ref: https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/ - primary: - extendedConfiguration: | - max_connections = 300 - - #PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: demand-devops - - # -- Select nodes to deploy which matches the following labels - nodeSelector: ##PLACEHOLDER## - dedicated: demand-devops - -# -- Provide a name in place of `aurva` -# namespaceOverride: aurva-dataplane - -########################################################## -# Global Configs -########################################################## -global: - aurva_controller: - enabled: true - aurva_fastdet: - enabled: false - aurva_pii_analyzer: - enabled: true - aurva_ocr: - enabled: true - aurva_collector: - enabled: true - - deploymentAnnotations: {} - - priorityClassName: "" - -########################################################## -# Aurva Controller -########################################################## -aurva_controller: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "demand" - team: "demand-devops" - service: "aurva-demand-prd" - env: "prd" - priority: "p0" - type: "aurva_controller" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 10 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - image: - # Image of the app container - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-controller/aurva-controller - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - - type: secret - name: aurva-controller-secrets - # - type: configmap - # name: proxy-datasource-config - - # -- Resources to be defined for pod - resources: - limits: - memory: 2.5Gi - cpu: 2 - requests: - memory: 2Gi - cpu: 1 - - aurvaFastdet: - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-fastdet/aurva-fastdet - tag: "v2.30.13" - pullPolicy: IfNotPresent - envFrom: [] - env: [] - resources: - limits: - cpu: 0.5 - memory: 512Mi - requests: - cpu: 0.5 - memory: 512Mi - - nodeSelector: ##PLACEHOLDER## - dedicated: demand-devops - - # -- Taint tolerations for nodes - ##PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: demand-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-controller-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - config: - #variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - COMMAND_URL: "command.aurva-prd.meeshogcp.in:80" - DEPLOYMENT_TYPE: "kubernetes" - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - FLUSHER_WORKER_POOL_SIZE: "1000" - FLUSHER_BATCH_SIZE: "30000" - UNIQUENESS_IDENTIFIER: "k8s-demand-prd-ase1" #Recommendation: should be equal to cluster name - PROVIDER_ACCOUNT_ID: "meesho-demand-prd-0622" # GCP PROJECT ID (not Number) - REGION: "asia-southeast1" #eg: asia-south1 - ENVIRONMENT: "prod" - OCR_ENABLED: "true" - MONITORING_ENABLED: "false" - HYBRID_ONLY_MODE: "false" - FORCE_TLS: "false" - FASTDET_FLAG : "true" - AADHAAR_ENHANCER: "0" - WORKSPACE_EVENT_TRACKING_ENABLED: "false" - ACCESS_IQ_ENABLED: "true" - GRPC_ENFORCE_ALPN_ENABLED: "false" - ENABLE_FASTDET: "true" - FASTDET_MAX_BATCH_SIZE: "100" - HEARTBEAT_INTERVAL: "5m" - #pii data - PII_BUCKET_NAME: "gcs-infra-devop-aurva-demand-prd" - PII_LOG_BUCKET_REGION: "asia-southeast1" - PII_LOG_CRON: "*/5 * * * *" - ENABLE_PII_LOG: "true" - PII_EVIDENCE_UPLOAD_MAX_BLOCKING_TASKS: "50000" - PII_EVIDENCE_UPLOAD_MAX_CONCURRENT_TASKS: "500" - PII_EVIDENCE_MAX_CACHE_WEIGHT: "100" - QUOTA_CLEANUP_CRON: "0 0 * * *" - ENABLE_PII_QUOTA: "true" - MAX_PII_EVIDENCES_PER_KEY_PER_WINDOW: "3" - PII_QUOTA_SYNC_CRON: "*/2 * * * *" - QUOTA_WINDOW_HOURS: "24" - #constants - SKIP_NAMESPACES: "contour-internal-1-demand-prd,contour-internal-0-demand-prd,contour-external-demand-prd" - CLOUD_PROVIDER: "gcp" - LOG_ENV: "production" - RDS_SCANNER_AVAILABILITY : "false" - REDSHIFT_SCANNER_AVAILABILITY : "false" - S3_SCANNER_AVAILABILITY : "false" - DYNAMO_SCANNER_AVAILABILITY: "false" - DOCDB_SCANNER_AVAILABILITY: "false" - OPENSEARCH_SCANNER_AVAILABILITY: "false" - CLOUDSQL_SCANNER_AVAILABILITY: "true" - BIGQUERY_SCANNER_AVAILABILITY: "true" - AWS_SNAPSHOT_SCANNER_AVAILABILITY: "false" - CLOUDSTORAGE_SCANNER_AVAILABILITY: "true" - KEYSPACES_SCANNER_AVAILABILITY: "false" - ALLOYDB_SCANNER_AVAILABILITY: "true" - BIGTABLE_SCANNER_AVAILABILITY: "true" - GCP_BACKUP_AVAILABILITY: "true" - EGRESS_MODE_ONLY: "false" - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - SCAN_UUID_ENABLED: "true" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-controller-sa - # -- Annotations applied to created service account - annotations: - iam.gke.io/gcp-service-account: sa-demand-prd-aurva-controller@meesho-demand-prd-0622.iam.gserviceaccount.com -# eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 15 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 - - -########################################################## -# Aurva OCR -########################################################## -aurva_ocr: - # -- Additional labels for aurva-controller - additionalLabels: - bu: "demand" - team: "demand-devops" - service: "aurva-demand-prd" - env: "prd" - priority: "p0" - type: "aurva_ocr" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 1 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-ocr/aurva-ocr - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-ocr: - type: secret - name: aurva-ocr-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 1 - requests: - memory: 2Gi - cpu: 1 - - # -- Select nodes to deploy which matches the following labels - - nodeSelector: ##PLACEHOLDER## - dedicated: demand-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: demand-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-ocr-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - OCR_TIME_LIMIT: "1" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-demand-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-ocr-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # iam.gke.io/gcp-service-account: service-account@gcp.iam.gserviceaccount.com - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva Collector -########################################################## -aurva_collector: - - # -- Additional labels for aurva-analyzer - additionalLabels: - bu: "demand" - team: "demand-devops" - service: "aurva-demand-prd" - env: "prd" - priority: "p0" - type: "aurva_collector" - - - # -- Annotations on aurva-analyzer - annotations: {} - # "key": "value" - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-collector/aurva-collector - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-controller: - type: secret - name: aurva-collector-secrets - - resources: - limits: - cpu: 800m - memory: 800Mi - requests: - cpu: 200m - memory: 512Mi - - podSecurityContext: {} - - securityContext: - privileged: true - capabilities: - add: - # For kernel v5.8 and above we don't need CAP_SYS_ADMIN or CAP_SYS_RESOURCE - # we just need CAP_BPF and CAP_PERFMON. This has been tested on our EKS node - # which is on kernel v5.10.x - # When SSL Tracing is required we need CAP_SYS_ADMIN and CAP_SYS_PTRACE - # on top of the previous capabilities - # So finally these are the 4 possible combinations for capabilies - # 1. Newer Kernels without SSL - # - BPF - # - PERFMON - # 2. Newer Kernels with SSL - - SYS_ADMIN - - SYS_PTRACE - # 3. Older Kernels without SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # 4. Older Kernels with SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # - SYS_PTRACE - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - volumes: - - name: debugfs - mountPath: /sys/kernel/debug - hostPath: /sys/kernel/debug - - name: vmlinux - mountPath: /sys/kernel/btf/vmlinux - hostPath: /sys/kernel/btf/vmlinux - - name: procfs - mountPath: /host/proc - hostPath: /proc - - name: bpffs - mountPath: /sys/fs/bpf - hostPath: /sys/fs/bpf - - # -- Taint tolerations for nodes - tolerations: - # - effect: NoSchedule - # key: dedicated - # operator: Equal - # value: megatetra - - operator: Exists - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external - - alloy - - preprod-spot-16 - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-collector-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - # variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-demand-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - TRACE_INTERNAL_SVC: "true" - TRACE_INTERNAL_SVC_HTTP: "true" - INTERNAL_SVC_SAMPLE_INTERVAL: "10m" - LOGS_TTL: "1h" - TRACE_HTTP2: "true" - TRACE_SSL: "false" - TRACE_PSQL: "false" - TRACE_SQLSERVER: "false" - TRACE_MYSQL: "false" - TRACE_EGRESS: "true" - TRACE_GO_TLS: "false" - TRACE_ML_SERVICES: "false" - # constants - LOG_ENV: production - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - MONITORING_ENABLED: "false" - ENABLE_INGRESS_INFORMER: "false" - ENABLE_SERVICE_INFORMER: "false" - ENABLE_ISTIO_INFORMER: "false" - EXCLUDED_PII_REGEX_TYPES: "ip_address,us_bank_number,us_driver_license,us_itin,us_passport,us_routing,us_mbi,ssn" - AGGREGATOR_MAX_CONNECTIONS: "1000" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-collector-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva PII Analyzer -########################################################## -aurva_pii_analyzer: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "demand" - team: "demand-devops" - service: "aurva-demand-prd" - env: "prd" - priority: "p0" - type: "aurva_pii_analyzer" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 3 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - ##PLACEHOLDER## - nodeSelector: - dedicated: demand-devops - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-piianalyzer/aurva-piianalyzer - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-pii-analyzer: - type: secret - name: aurva-pii-analyzer-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 4Gi - cpu: 4 - requests: - memory: 2Gi - cpu: 2 - - nodeSelector: ##PLACEHOLDER## - dedicated: demand-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: demand-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-pii-analyzer-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - SCHEDULER_TIME: "1" - SUPPORTED_REGION: "US" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-demand-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-pii-analyzer-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 3 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 diff --git a/helm-overrides/k8s-demand-prd-ase1/cert-manager/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/cert-manager/custom-values.yaml deleted file mode 100644 index 0e3db02..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/cert-manager/custom-values.yaml +++ /dev/null @@ -1,129 +0,0 @@ -global: - logLevel: 2 - rbac: - create: true - priorityClassName: "high-priority" -installCRDs: false - -crds: - enabled: true - keep: true - -# Cert-manager Controller -replicaCount: 1 -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-controller - tag: v1.20.1 - pullPolicy: IfNotPresent - -nodeSelector: - dedicated: demand-devops - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "demand-devops" - effect: "NoSchedule" - -resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 500m - memory: 512Mi - -# Webhook Configuration -webhook: - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-webhook - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: demand-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "demand-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# CA Injector Configuration -cainjector: - enabled: true - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-cainjector - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: demand-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "demand-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# ACME Solver Configuration -acmesolver: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-acmesolver - tag: v1.20.1 - pullPolicy: IfNotPresent - -# Startup API Check -startupapicheck: - enabled: true - timeout: 1m - backoffLimit: 4 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-startupapicheck - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: demand-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "demand-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - -# Prometheus Monitoring -prometheus: - enabled: true - servicemonitor: - enabled: false - interval: 60s - scrapeTimeout: 30s - labels: - prometheus: cert-manager diff --git a/helm-overrides/k8s-demand-prd-ase1/conntrack-adjuster/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index c390837..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,20 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - max: 2097152 - hashsize: 524288 - sleepInterval: 30 - -nodeAffinity: - enabled: true - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: In - values: - - "contour-external" - - "contour-internal-0" - - "contour-internal-1" - - "contour-intra-0" - - "contour-intra-1" diff --git a/helm-overrides/k8s-demand-prd-ase1/contour-ca-issuer/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/contour-ca-issuer/custom-values.yaml deleted file mode 100644 index 65b056e..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/contour-ca-issuer/custom-values.yaml +++ /dev/null @@ -1,17 +0,0 @@ -# ClusterIssuer override for k8s-demand-prd-ase1 (prd demand cluster, ase1a zone). -# ClusterIssuer is a cluster-scoped resource — this file pins the issuer -# identity for this cluster so the name is auditable per-cluster. -# -# Must match the `issuerRef.name` in consuming Certificate CRs (see -# devops-helm-charts/2.0.0/templates/proxyless-grpc-cert.yaml). - -issuerName: contour-demand-prd-ca-issuer -rootCASecretName: contour-demand-ca - -externalSecret: - enabled: true - vaultPath: meesho/devops/contour/root-ca - refreshInterval: "0" - secretStoreRef: - name: vault-backend -namespace: cert-manager-demand-prd \ No newline at end of file diff --git a/helm-overrides/k8s-demand-prd-ase1/contour-cert-checker/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/contour-cert-checker/custom-values.yaml deleted file mode 100644 index de1a22c..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=k8s-demand-prd-ase1"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - dedicated: demand-devops - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: demand-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/k8s-demand-prd-ase1/contour-external/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/contour-external/custom-values.yaml deleted file mode 100644 index fb0b71d..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/contour-external/custom-values.yaml +++ /dev/null @@ -1,85 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - network: - num-trusted-hops: 1 -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: demand - team: demand-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external - nodeSelector: - dedicated: contour-external - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: demand - team: demand-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external - nodeSelector: - dedicated: contour-external - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "40" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 2500m - memory: 512Mi - service: - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-demand-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-demand-prd-ase1/contour-internal-0/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/contour-internal-0/custom-values.yaml deleted file mode 100644 index 13e52db..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,92 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: demand - team: demand-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 4Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-shared - nodeSelector: - dedicated: contour-shared - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - updateStrategy: - rollingUpdate: - maxSurge: 10% - enabled: true - podLabels: - bu: demand - team: demand-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 350 - targetCPU: "40" - targetMemory: "55" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 12 - memory: 8Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int-0-demand-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-demand-prd-ase1/contour-internal-1/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/contour-internal-1/custom-values.yaml deleted file mode 100644 index 0969289..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,89 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: demand - team: demand-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 8 - memory: 10Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: demand - team: demand-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 350 - targetCPU: "40" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int-1-demand-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-demand-prd-ase1/contour-internal-intra-0/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index ee70ab0..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,90 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: demand - team: demand-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 8 - memory: 10Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - updateStrategy: - rollingUpdate: - maxSurge: 10% - enabled: true - podLabels: - bu: demand - team: demand-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-intra-0 - nodeSelector: - dedicated: contour-intra-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 350 - targetCPU: "40" - targetMemory: "55" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 12 - memory: 8Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-demand-prd-ase1/contour-internal-intra-1/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index 34796eb..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,87 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: demand - team: demand-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 12 - memory: 22Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: demand - team: demand-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-intra-1 - nodeSelector: - dedicated: contour-intra-1 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 350 - targetCPU: "40" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-demand-prd-ase1/coredns/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/coredns/custom-values.yaml deleted file mode 100644 index 33dd51f..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/coredns/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -replicaCount: 72 -communicationType: "intra" - -labels: - bu: demand - team: demand-devops - env: prd - -clusterIP: 10.137.0.2 - - -resources: - limits: - cpu: 150m - memory: 128Mi - requests: - cpu: 150m - memory: 128Mi - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "demand-devops" - effect: "NoSchedule" - -nodeSelector: - dedicated: demand-devops - kubernetes.io/os: linux diff --git a/helm-overrides/k8s-demand-prd-ase1/coroot-node-agent/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/coroot-node-agent/custom-values.yaml deleted file mode 100644 index ac390de..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/coroot-node-agent/custom-values.yaml +++ /dev/null @@ -1,52 +0,0 @@ -fullnameOverride: "coroot-node-agent-demand-prd" -priorityClassName: "low-priority" - -scrape: "false" # disable scraping for now - -resources: - requests: - cpu: "300m" - memory: "1Gi" - limits: - cpu: "500m" - memory: "1.5Gi" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-demand-contour-ext - operator: NotIn - values: - - dedicated - - key: p-demand-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-demand-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-demand-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-demand-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external \ No newline at end of file diff --git a/helm-overrides/k8s-demand-prd-ase1/external-secrets/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/external-secrets/custom-values.yaml deleted file mode 100644 index c152025..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/external-secrets/custom-values.yaml +++ /dev/null @@ -1,61 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Annotations to add to Pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: demand-devops - nodeSelector: - dedicated: demand-devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: demand-devops - nodeSelector: - dedicated: demand-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: demand-devops - nodeSelector: - dedicated: demand-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" \ No newline at end of file diff --git a/helm-overrides/k8s-demand-prd-ase1/flagger/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/flagger/custom-values.yaml deleted file mode 100644 index 0e98478..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/flagger/custom-values.yaml +++ /dev/null @@ -1,62 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - dedicated: demand-devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: demand-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: demand-devops - bu: infra - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-demand-rollout-service.prd-demand-rollout-service.svc.cluster.local" - -rolloutService: - enabled: true - apps: \ No newline at end of file diff --git a/helm-overrides/k8s-demand-prd-ase1/fluentd/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/fluentd/custom-values.yaml deleted file mode 100644 index 564cf89..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/fluentd/custom-values.yaml +++ /dev/null @@ -1,734 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-dmnd-dmsre-fluentd-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: demand - team: sre - type: fluentd - service: fluentd-demand-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: demand - team: sre - type: fluentd - service: fluentd-demand-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: - - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - from_encoding "utf-8" - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-demand-prd-ase1/keda/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/keda/custom-values.yaml deleted file mode 100644 index 9bb1d5a..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/keda/custom-values.yaml +++ /dev/null @@ -1,61 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: demand-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "demand-devops" - effect: "NoSchedule" - podLabels: - bu: "demand" - team: "demand-devops" - metricsAdapter: - bu: "demand" - team: "demand-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 1600Mi - requests: - cpu: 500m - memory: 850Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 60m - memory: 200Mi - metricServer: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 200m - memory: 200Mi - podAnnotations: - # -- Pod annotations for KEDA operator - keda: - prometheus.io/path: /metrics - prometheus.io/port: '8080' - prometheus.io/scrape: 'true' - # -- Pod annotations for KEDA Metrics Adapter - metricsAdapter: - prometheus.io/path: /metrics - prometheus.io/port: '8080' - prometheus.io/scrape: 'true' - # -- Pod annotations for KEDA Admission webhooks - webhooks: - prometheus.io/path: /metrics - prometheus.io/port: '8080' - prometheus.io/scrape: 'true' - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' diff --git a/helm-overrides/k8s-demand-prd-ase1/kube-dns/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/kube-dns/custom-values.yaml deleted file mode 100644 index a8f6aa9..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.137.0.2"],"prd.mrouter.int.svc.cluster.local":["10.137.0.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.137.0.2"]} diff --git a/helm-overrides/k8s-demand-prd-ase1/kube-events/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/kube-events/custom-values.yaml deleted file mode 100644 index 4bd0386..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-demand-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - dedicated: demand-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: demand-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: demand-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: demand-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: demand-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: demand-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-demand-prd-ase1/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 7cff8b2..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-demand-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-demand-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "demand" - team: "demand-sre" - service: "kube-state-metrics-demand-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "vmselect-mds" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 17m - memory: 1.5Gi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-demand-prd-ase1/kubectl-mcp-server/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/kubectl-mcp-server/custom-values.yaml deleted file mode 100644 index 8f29024..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/kubectl-mcp-server/custom-values.yaml +++ /dev/null @@ -1,117 +0,0 @@ -fullnameOverride: "kubectl-mcp-server" - -replicas: 1 - -image: - # In-house hardened build (helm-templates/kubectl-mcp-server/docker/) — - # NOT the upstream Docker Hub image. 266→18 HIGH/CRIT vulns, non-root, - # multi-stage slim base, kubectl v1.33.12 / helm v3.21.0. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/prd/devop/kubectl-mcp-server - tag: "v2" - pullPolicy: IfNotPresent - -labels: - bu: infra - team: devops - service: kubectl-mcp-server - env: prd - -serviceAccount: - create: true - annotations: {} - -rbac: - create: true - -podAnnotations: {} - -podSecurityContext: {} - -securityContext: - runAsNonRoot: true - runAsUser: 1000 - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -priorityClassName: "" - -nodeSelector: - dedicated: "demand-devops" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "demand-devops" - effect: NoSchedule - -affinity: {} - -resources: - requests: - cpu: 250m - memory: 512Mi - limits: - cpu: 1000m - memory: 1Gi - -mcp: - mode: single - transport: http - host: "0.0.0.0" - port: 8000 - -auth: - allowAnonymous: false - -externalSecrets: - enabled: true - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/prd/cntr/devop/kubectl-mcp-server-demand" - -# tcpSocket on purpose — streamable-http transport exposes only /mcp, -# there is no /health route (see chart values.yaml note). -livenessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - -readinessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - -service: - port: 8000 - type: ClusterIP - -# Contour standard (2.0.0-style): renders HTTPProxy parent/child + -intra -# instead of a networking.k8s.io Ingress. -createContourGateway: true - -ingress: - enabled: true - ingressClassName: contour-internal-1 - servicePortNumber: 8000 - hosts: - - host: kubectl-mcp-server-demand.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 diff --git a/helm-overrides/k8s-demand-prd-ase1/kyverno/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/kyverno/custom-values.yaml deleted file mode 100644 index bcf34d3..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/kyverno/custom-values.yaml +++ /dev/null @@ -1,2240 +0,0 @@ -global: - - # -- Internal settings used with `helm template` to generate install manifest - # @ignored - templating: - enabled: false - debug: false - version: ~ - - image: - # -- (string) Global value that allows to set a single image registry across all deployments. - # When set, it will override any values set under `.image.registry` across the chart. - registry: ~ - # -- (list) Global list of Image pull secrets - # When set, it will override any values set under `imagePullSecrets` under different components across the chart. - imagePullSecrets: [] - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - caCertificates: - # -- Global CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - # Individual controller values will override this global value - data: ~ - - # -- Global value to set single volume to be mounted for CA certificates for all deployments. - # Not used when `.Values.global.caCertificates.data` is defined - # Individual controller values will override this global value - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Additional container environment variables to apply to all containers and init containers - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Global node labels for pod assignment. Non-global values will override the global value. - nodeSelector: - dedicated: demand-kyverno - - # -- Global List of node taints to tolerate. Non-global values will override the global value. - tolerations: - - key: "dedicated" - operator: "Equal" - value: "demand-kyverno" - effect: "NoSchedule" - -# -- (string) Override the name of the chart -nameOverride: ~ - -# -- (string) Override the expanded name of the chart -fullnameOverride: ~ - -# -- (string) Override the namespace the chart deploys to -namespaceOverride: ~ - -upgrade: - # -- Upgrading from v2 to v3 is not allowed by default, set this to true once changes have been reviewed. - fromV2: false - -apiVersionOverride: - # -- (string) Override api version used to create `PodDisruptionBudget`` resources. - # When not specified the chart will check if `policy/v1/PodDisruptionBudget` is available to - # determine the api version automatically. - podDisruptionBudget: ~ - -rbac: - roles: - # -- Aggregate ClusterRoles to Kubernetes default user-facing roles. For more information, see [User-facing roles](https://kubernetes.io/docs/reference/access-authn-authz/rbac/#user-facing-roles) - aggregate: - admin: true - view: true - -# Use openreports.io as the API group for reporting -openreports: - # -- Enable OpenReports feature in controllers - enabled: false - # -- Whether to install CRDs from the upstream OpenReports chart. Setting this to true requires enabled to also be true. - installCrds: false - -# CRDs configuration -crds: - - # -- Whether to have Helm install the Kyverno CRDs, if the CRDs are not installed by Helm, they must be added before policies can be created - install: true - - reportsServer: - # -- Kyverno reports-server is used in your cluster - enabled: false - - groups: - - # -- Install CRDs in group `kyverno.io` - kyverno: - cleanuppolicies: true - clustercleanuppolicies: true - clusterpolicies: true - globalcontextentries: true - policies: true - policyexceptions: true - updaterequests: true - - # -- Install CRDs in group `policies.kyverno.io` - policies: - validatingpolicies: true - policyexceptions: true - imagevalidatingpolicies: true - namespacedimagevalidatingpolicies: true - mutatingpolicies: true - generatingpolicies: true - deletingpolicies: true - namespaceddeletingpolicies: true - namespacedvalidatingpolicies: true - - # -- Install CRDs in group `reports.kyverno.io` - reports: - clusterephemeralreports: true - ephemeralreports: true - - # -- Install CRDs in group `wgpolicyk8s.io` - wgpolicyk8s: - clusterpolicyreports: false ##lk - policyreports: false ##lk - - # -- Additional CRDs annotations (Replace=true avoids 262144-byte annotation limit on large CRDs) - annotations: - argocd.argoproj.io/sync-options: Replace=true - # strategy.spinnaker.io/replace: 'true' - - # -- Additional CRDs labels - customLabels: {} - - migration: - - # -- Enable CRDs migration using helm post upgrade hook - enabled: true - - # -- Resources to migrate - resources: - - cleanuppolicies.kyverno.io - - clustercleanuppolicies.kyverno.io - - clusterpolicies.kyverno.io - - globalcontextentries.kyverno.io - - policies.kyverno.io - - policyexceptions.kyverno.io - - updaterequests.kyverno.io - - deletingpolicies.policies.kyverno.io - - generatingpolicies.policies.kyverno.io - - imagevalidatingpolicies.policies.kyverno.io - - namespacedimagevalidatingpolicies.policies.kyverno.io - - mutatingpolicies.policies.kyverno.io - - namespaceddeletingpolicies.policies.kyverno.io - - namespacedvalidatingpolicies.policies.kyverno.io - - policyexceptions.policies.kyverno.io - - validatingpolicies.policies.kyverno.io - - image: - # -- (string) Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- (string) Image repository - repository: kyverno/kyverno-cli - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- (string) Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podResources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -# Configuration -config: - - # -- Create the configmap. - create: true - - # -- Preserve the configmap settings during upgrade. - preserve: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - # -- Enable registry mutation for container images. Enabled by default. - enableDefaultRegistryMutation: true - - # -- The registry hostname used for the image mutation. - defaultRegistry: docker.io - - # -- Exclude groups - excludeGroups: - - system:nodes - - # -- Exclude usernames - excludeUsernames: [] - # - '!system:kube-scheduler' - - # -- Exclude roles - excludeRoles: [] - - # -- Exclude roles - excludeClusterRoles: [] - - # -- Generate success events. - generateSuccessEvents: false - - # -- Resource types to be skipped by the Kyverno policy engine. - # Make sure to surround each entry in quotes so that it doesn't get parsed as a nested YAML list. - # These are joined together without spaces, run through `tpl`, and the result is set in the config map. - # @default -- See [values.yaml](values.yaml) - resourceFilters: - - '[Event,*,*]' - # - '[*/*,kube-system,*]' - - '[*/*,kube-public,*]' - - '[*/*,kube-node-lease,*]' - - '[Node,*,*]' - - '[Node/?*,*,*]' - - '[APIService,*,*]' - - '[APIService/?*,*,*]' - - '[TokenReview,*,*]' - - '[SubjectAccessReview,*,*]' - - '[SelfSubjectAccessReview,*,*]' - - '[Binding,*,*]' - - '[Pod/binding,*,*]' - - '[ReplicaSet,*,*]' - - '[ReplicaSet/?*,*,*]' - - '[EphemeralReport,*,*]' - - '[ClusterEphemeralReport,*,*]' - # exclude resources from the chart - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:additional]' - - '[ClusterRoleBinding,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.configMapName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.metricsConfigMapName" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Job,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[Job/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.admission-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.background-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - # -- Sets the threshold for the total number of UpdateRequests generated for mutateExisitng and generate policies. - updateRequestThreshold: 1000 - - # -- Defines the `namespaceSelector`/`objectSelector` in the webhook configurations. - # The Kyverno namespace is excluded if `excludeKyvernoNamespace` is `true` (default) - webhooks: - # Exclude namespaces - # namespaceSelector: lk-s2 - # matchExpressions: - # - key: kubernetes.io/metadata.name - # operator: NotIn - # values: - # - kube-systemm lk-s2 - # Exclude objects - # objectSelector: - # matchExpressions: - # - key: webhooks.kyverno.io/exclude - # operator: DoesNotExist - - # -- Defines annotations to set on webhook configurations. - webhookAnnotations: - # Example to disable admission enforcer on AKS: - # 'admissions.enforcer/disabled': 'true' lk-s3 - - # -- Defines labels to set on webhook configurations. - webhookLabels: {} - # Example to adopt webhook resources in ArgoCD: - # 'argocd.argoproj.io/instance': 'kyverno' - - # -- Defines match conditions to set on webhook configurations (requires Kubernetes 1.27+). - matchConditions: [] - - # -- Exclude Kyverno namespace - # Determines if default Kyverno namespace exclusion is enabled for webhooks and resourceFilters - excludeKyvernoNamespace: true - - # -- resourceFilter namespace exclude - # Namespaces to exclude from the default resourceFilters - resourceFiltersExcludeNamespaces: [] - - # -- resourceFilters exclude list - # Items to exclude from config.resourceFilters - resourceFiltersExclude: [] - - # -- resourceFilter namespace include - # Namespaces to include to the default resourceFilters - resourceFiltersIncludeNamespaces: [] - - # -- resourceFilters include list - # Items to include to config.resourceFilters - resourceFiltersInclude: [] - -# Metrics configuration -metricsConfig: - - # -- Create the configmap. - create: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - namespaces: - - # -- List of namespaces to capture metrics for. - include: [] - - # -- list of namespaces to NOT capture metrics for. - exclude: [] - - # -- (string) Rate at which metrics should reset so as to clean up the memory footprint of kyverno metrics, if you might be expecting high memory footprint of Kyverno's metrics. Default: 0, no refresh of metrics. WARNING: This flag is not working since Kyverno 1.8.0 - metricsRefreshInterval: ~ - # metricsRefreshInterval: 24h - - # -- (list) Configures the bucket boundaries for all Histogram metrics, changing this configuration requires restart of the kyverno admission controller - bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 15, 20, 25, 30] - - # -- (map) Configures the exposure of individual metrics, by default all metrics and all labels are exported, changing this configuration requires restart of the kyverno admission controller - metricsExposure: - kyverno_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_image_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_mutating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_generating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_admission_review_duration_seconds: - # enabled: false - disabledLabelDimensions: ["resource_namespace"] - kyverno_policy_rule_info_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_policy_results_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_admission_requests_total: - disabledLabelDimensions: ["resource_namespace"] - kyverno_cleanup_controller_deletedobjects_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - -# -- Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -imagePullSecrets: {} - # regcred: - # registry: foo.example.com - # username: foobar - # password: secret - # regcred2: - # registry: bar.example.com - # username: barbaz - # password: secret2 - -# -- Existing Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -existingImagePullSecrets: [] - # - test-registry - # - other-test-registry - -# Tests configuration -test: - # -- Sleep time before running test - sleep: 20 - - image: - # -- (string) Image registry - registry: curlimages - # -- Image repository - repository: curl - # -- Image tag - # Defaults to `latest` if omitted - tag: '8.10.1' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - # -- Security context for the test containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- Additional Pod annotations - podAnnotations: {} - - # -- List of node taints to tolerate - tolerations: [] - -# -- Additional labels -customLabels: {} - - -webhooksCleanup: - # -- Create a helm pre-delete hook to cleanup webhooks. - enabled: true - - autoDeleteWebhooks: - # -- Allow webhooks controller to delete webhooks using finalizers - enabled: false - - image: - # -- (string) Image registry - registry: registry.k8s.io - # -- Image repository - repository: kubectl - # -- Image tag - # Defaults to `latest` if omitted - tag: 'v1.32.7' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -grafana: - # -- Enable grafana dashboard creation. - enabled: false - - # -- Configmap name template. - configMapName: '{{ include "kyverno.fullname" . }}-grafana' - - # -- (string) Namespace to create the grafana dashboard configmap. - # If not set, it will be created in the same namespace where the chart is deployed. - namespace: ~ - - # -- Grafana dashboard configmap annotations. - annotations: {} - - # -- Grafana dashboard configmap labels - labels: - grafana_dashboard: "1" - - # -- create GrafanaDashboard custom resource referencing to the configMap. - # according to https://grafana-operator.github.io/grafana-operator/docs/examples/dashboard_from_configmap/readme/ - grafanaDashboard: - create: false - folder: kyverno - allowCrossNamespaceImport: true - matchLabels: - dashboards: "grafana" - -# Features configuration -features: - admissionReports: - # -- Enables the feature - enabled: true - aggregateReports: - # -- Enables the feature - enabled: true - policyReports: - # -- Enables the feature - enabled: false ##lk - validatingAdmissionPolicyReports: - # -- Enables the feature - enabled: true - mutatingAdmissionPolicyReports: - # -- Enables the feature - enabled: false - reporting: - # -- Enables the feature - validate: true - # -- Enables the feature - mutate: true - # -- Enables the feature - mutateExisting: true - # -- Enables the feature - imageVerify: true - # -- Enables the feature - generate: true - autoUpdateWebhooks: - # -- Enables the feature - enabled: true - backgroundScan: - # -- Enables the feature - enabled: true - # -- Number of background scan workers - backgroundScanWorkers: 2 - # -- Background scan interval - backgroundScanInterval: 1h - # -- Skips resource filters in background scan - skipResourceFilters: true - configMapCaching: - # -- Enables the feature - enabled: true - controllerRuntimeMetrics: - # -- Bind address for controller-runtime metrics (use "0" to disable it) - bindAddress: ":8080" - deferredLoading: - # -- Enables the feature - enabled: true - dumpPayload: - # -- Enables the feature - enabled: false - forceFailurePolicyIgnore: - # -- Enables the feature - enabled: false - generateValidatingAdmissionPolicy: - # -- Enables the feature - enabled: true - generateMutatingAdmissionPolicy: - # -- Enables the feature - enabled: false - dumpPatches: - # -- Enables the feature - enabled: false - globalContext: - # -- Maximum allowed response size from API Calls. A value of 0 bypasses checks (not recommended) - maxApiCallResponseLength: 2000000 - logging: - # -- Logging format - format: text - # -- Logging verbosity - verbosity: 2 - omitEvents: - # -- Events which should not be emitted (possible values `PolicyViolation`, `PolicyApplied`, `PolicyError`, and `PolicySkipped`) - eventTypes: - - PolicyApplied - - PolicySkipped - # - PolicyViolation - # - PolicyError - policyExceptions: - # -- Enables the feature - enabled: true - # -- Restrict policy exceptions to a single namespace - # Set to "*" to allow exceptions in all namespaces - namespace: '' - protectManagedResources: - # -- Enables the feature - enabled: false - registryClient: - # -- Allow insecure registry - allowInsecure: false - # -- Enable registry client helpers - credentialHelpers: - - default - - google - - amazon - - azure - - github - ttlController: - # -- Reconciliation interval for the label based cleanup manager - reconciliationInterval: 1m - tuf: - # -- Enables the feature - enabled: false - # -- (string) Path to Tuf root - root: ~ - # -- (string) Raw Tuf root - rootRaw: ~ - # -- (string) Tuf mirror - mirror: ~ - -# Admission controller configuration -admissionController: - autoscaling: - # -- Enable horizontal pod autoscaling - enabled: true - - # -- Minimum number of pods - minReplicas: 1 - - # -- Maximum number of pods - maxReplicas: 10 - - # -- Target CPU utilization percentage - targetCPUUtilizationPercentage: 80 - - # -- Configurable scaling behavior - behavior: {} - - # -- Overrides features defined at the root level - featuresOverride: - admissionReports: - # -- Max number of admission reports allowed in flight until the admission controller stops creating new ones - backPressureThreshold: 1000 - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- The ServiceAccount name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno admission controller activities. - # This will help ensure Kyverno stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- admissionController webhook server port - # in case you are using hostNetwork: true, you might want to change the port the webhookServer is listening to - webhookServer: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - admission-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.admissionController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - initContainer: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyvernopre - # -- (string) Image tag - # If missing, defaults to image.tag - tag: ~ - # -- (string) Image pull policy - # If missing, defaults to image.pullPolicy - pullPolicy: ~ - - resources: - # -- Pod resource limits - # limits: - # cpu: 100m - # memory: 256Mi - # -- Pod resource requests - requests: - cpu: 1000m - memory: 1Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - container: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyverno - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - resources: - # -- Pod resource limits - # limits: - # memory: 384Mi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Array of extra init containers - extraInitContainers: [] - # - name: init-container - # image: busybox - # command: ['sh', '-c', 'echo Hello'] - - # -- Array of extra containers to run alongside kyverno - extraContainers: [] - # - name: myapp-container - # image: busybox - # command: ['sh', '-c', 'echo Hello && sleep 3600'] - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Kyverno's metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Background controller configuration -backgroundController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable background controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: - - apiGroups: - - networking.k8s.io - resources: - - ingresses - - ingressclasses - - networkpolicies - verbs: - - create - - update - - patch - - delete - - apiGroups: - - rbac.authorization.k8s.io - resources: - - rolebindings - - roles - verbs: - - create - - update - - patch - - delete - - apiGroups: - - '' - resources: - - configmaps - - resourcequotas - - limitranges - verbs: - - create - - update - - patch - - delete - - apiGroups: - - resource.k8s.io - resources: - - resourceclaims - - resourceclaimtemplates - verbs: - - create - - delete - - update - - patch - - deletecollection - - apiGroups: - - autoscaling - resources: - - horizontalpodautoscalers - verbs: - - get - - list - - watch - - update - - patch - - apiGroups: - - apps - resources: - - deployments - - statefulsets - - replicasets - verbs: - - get - - list - - watch - - update - - patch - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - # - patch - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/background-controller - # -- Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - # limits: - # memory: 128Mi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - background-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.backgroundController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - # -- backgroundController server port - # in case you are using hostNetwork: true, you might want to change the port the backgroundController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Cleanup controller configuration -cleanupController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable cleanup controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - delete - # - list - # - watch - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/cleanup-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- cleanupController server port - # in case you are using hostNetwork: true, you might want to change the port the cleanupController is listening to - server: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - # limits: - # memory: 128Mi - # -- Pod resource requests - requests: - cpu: 1100m - memory: 1.2Gi - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - cleanup-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `service.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Reports controller configuration -reportsController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable reports controller. - enabled: false ##lk - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/reports-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno reports controller activities. - # This will help ensure Kyverno reports stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - memory: 128Mi - # -- Pod resource requests - requests: - cpu: 100m - memory: 64Mi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - reports-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.reportsController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- (string) Service node port. - # Only used if `type` is `NodePort`. - nodePort: ~ - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- (string) Traces receiver address - address: ~ - # -- (string) Traces receiver port - port: ~ - # -- (string) Traces receiver credentials - creds: ~ - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- (string) Otel collector endpoint - collector: ~ - # -- (string) Otel collector credentials - creds: ~ - - # -- reportsController server port - # in case you are using hostNetwork: true, you might want to change the port the reportsController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - - # -- Enable sanity check for reports CRDs - sanityChecks: true diff --git a/helm-overrides/k8s-demand-prd-ase1/kyverno/policies/protect-namespace.yaml b/helm-overrides/k8s-demand-prd-ase1/kyverno/policies/protect-namespace.yaml deleted file mode 100644 index 361e10c..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/kyverno/policies/protect-namespace.yaml +++ /dev/null @@ -1,40 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: protect-critical-namespaces -spec: - validationFailureAction: Enforce - background: false - rules: - - name: deny-deletion-of-critical-namespaces - match: - any: - - resources: - kinds: - - Namespace - names: - - keda-demand-prd - - contour-internal-0-demand-prd - - contour-internal-0-demand-prd-intra - - contour-external-demand-prd - - external-secrets-demand-prd - - flagger-demand-prd - - victoriametrics - - kube-system - - monitoring - - telegraf-operator - - loadtester - - opentelemetry - - fluentd - - kube-events - - observability - - contour-cert-checker-ns - - aurva-dataplane - validate: - message: "Deletion of critical namespaces is not allowed." - deny: - conditions: - any: - - key: "{{request.operation}}" - operator: Equals - value: DELETE diff --git a/helm-overrides/k8s-demand-prd-ase1/kyverno/policies/restrict-replicas.yaml b/helm-overrides/k8s-demand-prd-ase1/kyverno/policies/restrict-replicas.yaml deleted file mode 100644 index d456985..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/kyverno/policies/restrict-replicas.yaml +++ /dev/null @@ -1,29 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: restrict-minimum-replicas-selected-ns -spec: - validationFailureAction: Enforce - background: false - rules: - - name: min-replicas-deployments-statefulsets - match: - resources: - kinds: - - Deployment - - StatefulSet - - Deployment/scale - - StatefulSet/scale - names: - - coredns - namespaces: - - kube-system - operations: - - CREATE - - UPDATE - validate: - message: "Deployments and StatefulSets must have at least 5 replicas in this namespace." - anyPattern: - # Case 1: replicas explicitly set and >= 5 - - spec: - replicas: ">=5" diff --git a/helm-overrides/k8s-demand-prd-ase1/loadtester/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/loadtester/custom-values.yaml deleted file mode 100644 index 767754e..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/loadtester/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -replicaCount: 1 - -Namespace: loadtester - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/dev/devops/flagger-loadtester - tag: 0.30.0 - pullPolicy: IfNotPresent - pullSecret: - -podLabels: - bu: demand - team: demand-shared - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - -podPriorityClassName: "" - -logLevel: info -cmd: - timeout: 1h - namespaceRegexp: "" - -nameOverride: "loadtester" -fullnameOverride: "" - -env: [] - -service: - type: ClusterIP - port: 80 - -resources: - requests: - cpu: 10m - memory: 64Mi - -volumes: [] -volumeMounts: [] - -nodeSelector: - dedicated: demand-devops - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "demand-devops" - effect: "NoSchedule" - -affinity: {} - -ingress: - slowStart: - enabled: false - enabled: true - ingressClassName: contour-internal-1 - servicePort: http - hosts: - - host: demand-flagger-loadtester.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / -createContourGateway: true -contourResponseTimeout: false -rbac: - # rbac.create: `true` if rbac resources should be created - create: true - # rbac.scope: `cluster` to create cluster-scope rbac resources (ClusterRole/ClusterRoleBinding) - # otherwise, namespace-scope rbac resources will be created (Role/RoleBinding) - scope: - # rbac.rules: array of rules to apply to the role. example: - # rules: - # - apiGroups: [""] - # resources: ["pods"] - # verbs: ["list", "get"] - rules: [] - -# name of an existing service account to use - if not creating rbac resources -serviceAccountName: "" - -# App Mesh virtual node settings (to be used for AppMesh v1beta1) -meshName: "" -#backends: -# - app1.namespace -# - app2.namespace - -# App Mesh virtual node settings (to be used for AppMesh v1beta2) -appmesh: - enabled: false - backends: - - podinfo - - podinfo-canary - -#Istio virtual service and gatway settings. TLS secrets should be in namespace before enbaled it. ( secret format loadtester.fullname ) -istio: - enabled: false - host: flagger-loadtester.flagger - gateway: - enabled: false - tls: - enabled: false - httpsRedirect: false - -# when enabled, it will add a security context for the loadtester pod -securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 100 - runAsGroup: 101 - -podDisruptionBudget: - enabled: false - minAvailable: 1 diff --git a/helm-overrides/k8s-demand-prd-ase1/node-thp-config/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/node-thp-config/custom-values.yaml deleted file mode 100644 index a3e7460..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/node-thp-config/custom-values.yaml +++ /dev/null @@ -1,9 +0,0 @@ -daemonSet: - namespace: prd-node-thp-config - -baseMatchExpressions: -- key: dedicated - operator: In - values: - - "sumoduo-azul" - - "sumoduolite-azul" # change to your actual node label value diff --git a/helm-overrides/k8s-demand-prd-ase1/opentelemetry-coralogix/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/opentelemetry-coralogix/custom-values.yaml deleted file mode 100644 index 02b0132..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/opentelemetry-coralogix/custom-values.yaml +++ /dev/null @@ -1,281 +0,0 @@ -global: - domain: "coralogixsg.com" - defaultApplicationName: "default" - defaultSubsystemName: "nodes" - - # Old endpoint based configuration, - # please use domain instead. - traces: - endpoint: "" - metrics: - endpoint: "" - logs: - endpoint: "" - -# set distribution to openshift for openshift clusters -distribution: "" - -# Opentelemetry collector configuration -dedicatedValue: false -schedulerName: default-scheduler - -labels: - bu: "demand" - team: "sre" - service: "opentelemetry-demand-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: false - key: prd/admin/coralogix-keys - secretStoreRef: - name: vault-backend - -image: - # If you want to use the core image `otel/opentelemetry-collector`, you also need to change `command.name` value to `otelcol`. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - # repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - tag: "0.87.0" - # When digest is set to a non-empty value, images will be pulled by digest (regardless of tag value). - digest: "" -mode: daemonset -rollout: - rollingUpdate: - # When 'mode: daemonset', maxSurge cannot be used when hostPort is set for any of the ports - # maxSurge: 25% - maxUnavailable: 1 - strategy: RollingUpdate -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" -hostNetwork: true -dnsPolicy: "ClusterFirstWithHostNet" -fullnameOverride: opentelemetry-demand-prd - -presets: - logsCollection: - enabled: false - storeCheckpoints: false - maxRecombineLogSize: 1048576 - extraFilelogOperators: [] -# - type: recombine -# combine_field: body -# source_identifier: attributes["log.file.path"] -# is_first_entry: body matches "^(YOUR-LOGS-REGEX)" - kubernetesAttributes: - enabled: false - hostMetrics: - enabled: false - kubeletMetrics: - enabled: false - -extraEnvs: -- name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" -- name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName -config: - extensions: {} - # zpages: - # endpoint: localhost:55679 - # pprof: - # endpoint: localhost:1777 - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 4s - tls: - insecure: true - sending_queue: - enabled: true - queue_size: 6000 - num_consumers: 200 - resolver: - dns: - hostname: opentelemetry-admin-prd.opentelemetry.svc.clusterset.local - # coralogix: - # timeout: "30s" - # private_key: "${CORALOGIX_PRIVATE_KEY}" - # domain: "{{.Values.global.domain}}" - # traces: - # endpoint: "{{ .Values.global.traces.endpoint }}" - # metrics: - # endpoint: "{{ .Values.global.metrics.endpoint }}" - # logs: - # endpoint: "{{ .Values.global.logs.endpoint }}" - # application_name_attributes: - # - "k8s.namespace.name" - # - "service.namespace" - # subsystem_name_attributes: - # - "k8s.deployment.name" - # - "k8s.statefulset.name" - # - "k8s.daemonset.name" - # - "k8s.cronjob.name" - # - "k8s.job.name" - # - "k8s.container.name" - # - "k8s.node.name" - # - "service.name" - # application_name: "{{.Values.global.defaultApplicationName }}" - # subsystem_name: "{{.Values.global.defaultSubsystemName }}" - processors: - k8sattributes: - filter: - node_from_env_var: KUBE_NODE_NAME - extract: - metadata: - - "k8s.namespace.name" - - "k8s.deployment.name" - - "k8s.statefulset.name" - - "k8s.daemonset.name" - - "k8s.cronjob.name" - - "k8s.job.name" - - "k8s.pod.name" - - "k8s.node.name" - memory_limiter: null # Will get the k8s resource limits - resourcedetection/env: - detectors: ["system","env"] - timeout: 2s - override: false - # spanmetrics: - # metrics_exporter: coralogix - # dimensions: - # - name: "k8s.deployment.name" - # - name: "k8s.statefulset.name" - # - name: "k8s.daemonset.name" - # - name: "k8s.cronjob.name" - # - name: "k8s.job.name" - # - name: "k8s.container.name" - # - name: "k8s.node.name" - # - name: "k8s.namespace.name" - receivers: - otlp: - protocols: - grpc: - endpoint: ${MY_POD_IP}:4317 - http: - endpoint: ${MY_POD_IP}:4318 - zipkin: - endpoint: ${MY_POD_IP}:9411 - jaeger: - protocols: - grpc: - endpoint: ${MY_POD_IP}:14250 - thrift_http: - endpoint: ${MY_POD_IP}:14268 - thrift_compact: - endpoint: ${MY_POD_IP}:6831 - thrift_binary: - endpoint: ${MY_POD_IP}:6832 - prometheus: - config: - scrape_configs: - - job_name: opentelemetry-collector - scrape_interval: 30s - static_configs: - - targets: - - ${MY_POD_IP}:8888 - service: - extensions: - # - zpages - # - pprof - - health_check -# - memory_ballast - telemetry: - metrics: - address: ${env:MY_POD_IP}:8888 - pipelines: - traces: - exporters: - - loadbalancing - processors: - - memory_limiter - # - spanmetrics - - batch - receivers: - - otlp - metrics: null - logs: null -tolerations: - - operator: Exists - -resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 1 - memory: 2G - -ports: - jaeger-binary: - enabled: true - containerPort: 6832 - servicePort: 6832 - hostPort: 6832 - protocol: TCP - # In order to enable podMonitor, following part must be enabled in order to expose the required port: - # metrics: - # enabled: true - -# podMonitor: -# enabled: true - -# prometheusRule: -# enabled: true -# defaultRules: -# enabled: true - -# Annotations to be added to pod -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-demand-contour-ext - operator: NotIn - values: - - dedicated - - key: p-demand-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-demand-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-demand-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-demand-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert \ No newline at end of file diff --git a/helm-overrides/k8s-demand-prd-ase1/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index 02e1289..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,133 +0,0 @@ -fullnameOverride: opentelemetry-demand-prd - -mode: daemonset - -priorityClassName: "system-node-critical" - -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-demand-contour-ext - operator: NotIn - values: - - dedicated - - key: p-demand-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-demand-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-demand-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-demand-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 500m - memory: 256Mi - limits: - cpu: 1 - memory: 512Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "demand" - team: "sre" - service: "opentelemetry-demand-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/k8s-demand-prd-ase1/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index 56c4222..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,496 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 200m - # memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-admin-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "demand" - team: "demand-sre" - service: "node-exporter-demand-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-demand-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index 5f8e2c8..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-demand-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-demand-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "demand" - team: "demand-sre" - service: "stackdriver-exporter-demand-prd" - env: "prd" - priority: "p0" - type: "exporter" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-demand-prd-0622" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,kubernetes.io/node/ephemeral_storage/used_bytes,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime,compute.googleapis.com/instance' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - dedicated: vmselect-mds - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-dmnd-dmsre-stackdriver-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - } - - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/k8s-demand-prd-ase1/telegraf-operator/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/telegraf-operator/custom-values.yaml deleted file mode 100644 index a4befe3..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,292 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - - infra-histogram-expiration-enabled: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - expiration_interval = "2m" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - - infra-telegraf-cardinality-optimized: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - expiration_interval = "2m" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "demand" - team: "demand-sre" - service: "telegraf-operator-demand-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - dedicated: "vmselect-mds" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false \ No newline at end of file diff --git a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml deleted file mode 100644 index cbc3101..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml +++ /dev/null @@ -1,296 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-demand-prd-dr -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-demand-prd-dr-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent-dr - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dmnd-dmsre-vmagent-prd@meesho-demand-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - - https://vminsert-infra-prd-dr.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "demand" - team: "demand-sre" - service: "vmagent-demand-prd" - env: "prd" - priority: "p0" - type: "vmagent-dr" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-demand-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: metricsapi-demand-prd-dr.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 12 - memory: 10Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent-dr" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml deleted file mode 100644 index 198b5c2..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml +++ /dev/null @@ -1,272 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 1 - -fullnameOverride: vmagent-demand-prd-fb -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-demand-prd-fb-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - -# Remove topology spread constraints -topologySpreadConstraints: [] - -# Modify deployment strategy -deployment: - enabled: true - -# Remove HPA -horizontalPodAutoscaler: - enabled: false - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-infr-sre-vmagent-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -# Remove pod disruption budget -podDisruptionBudget: - enabled: false - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-demand.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vmsinglenode-infra-prd.victoriametrics.svc.clusterset.local:8428/api/v1/write - - https://vmsinglenode-infra-prd.meeshogcp.in/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "demand-fb" - team: "sre" - service: "vmagent-demand-prd-fb" - env: "prd" - priority: "p0" - type: "vmagent-fb" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-demand-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-demand-prd-fb.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -# Modify resources -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - requests: - cpu: 2 - memory: 3Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml deleted file mode 100644 index 43e4c2e..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert-dr - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-demand-prd-dr - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/demand/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-demand-prd-dr.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-demand-prd-dr" - env: "prd" - priority: "p0" - type: "vmalert-dr" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - dedicated: "vmagent-dr" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert-secured-stateful/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert-secured-stateful/custom-values.yaml deleted file mode 100644 index 46c52e6..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert-secured-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-stateful-secured-demand-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-demand-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-demand-prd.victoriametrics.svc.cluster.local:8480/insert/200/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-demand-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/sensitive/demand/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: false - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-demand-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "demand" - team: "sre" - service: "vmalert-stateful-secured-demand-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-secured-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert-secured/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert-secured/custom-values.yaml deleted file mode 100644 index b67417e..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert-secured/custom-values.yaml +++ /dev/null @@ -1,305 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-secured-demand-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-demand-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-demand-prd.victoriametrics.svc.cluster.local:8480/insert/200/prometheus" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-demand-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.cluster.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config1/sensitive/demand/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-secured-demand-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 1 - memory: 2Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "supply" - team: "sre" - service: "vmalert-secured-demand-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml deleted file mode 100644 index 93dc682..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-demand-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-demand-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-demand-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-demand-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/demand/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-demand-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 1500m - memory: 3Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "demand" - team: "sre" - service: "vmalert-demand-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert/custom-values.yaml deleted file mode 100644 index 8166388..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-alert/custom-values.yaml +++ /dev/null @@ -1,305 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-demand-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-demand-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-demand-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-demand-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config1/demand/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-demand-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-demand-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-insert/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index 8e182a2..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,234 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-demand-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-demand-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "demand" - team: "demand-sre" - service: "vminsert-demand-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 15 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 7 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vminsert-mds" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 10 - memory: 10Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-demand-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-select/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index c3a2eb2..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,293 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-demand-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-demand-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 384 - clusternative.maxConcurrentRequests: 384 - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "demand" - team: "demand-sre" - service: "vmselect-demand-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 15 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 7 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "vmselect-mds" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 36 - memory: 60Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: contour-internal-0 - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-demand.prd.meesho.int - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-storage/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index 61c4a58..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,316 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-demand-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage-sale-24aug" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage-sale-24aug" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 700Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "demand" - team: "demand-sre" - service: "vmstorage-demand-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 32 - memory: 600Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.93.7-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-demand-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-demand-prd-ase1/victoriametrics-agent-secondary/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/victoriametrics-agent-secondary/custom-values.yaml deleted file mode 100644 index da4488a..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/victoriametrics-agent-secondary/custom-values.yaml +++ /dev/null @@ -1,479 +0,0 @@ -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "demand-scrape-secondary.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -# -- Replica count -replicaCount: 2 - -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - -# -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables -deployment: - spec: - minReadySeconds: 1200 - progressDeadlineSeconds: 1800 - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - -image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - # -- Image tag, set to `Chart.AppVersion` by default - tag: v1.133.0 # rewrites Chart.AppVersion - # -- Variant of the image to use. - # e.g. enterprise, scratch - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - -# -- Image pull secrets -imagePullSecrets: [] - -# -- Add additional DNS entries to pods hosts file. Check [official documentation](https://kubernetes.io/docs/tasks/network/customize-hosts-file-for-pods/) -hostAliases: [] -# - ip: 192.168.1.1 -# hostNames: -# - test.example.com -# - another.example.net - -# -- Override chart name -nameOverride: "" - -# -- Override resources fullname -fullnameOverride: "vm-agent-demand-prd-secondary" - -# -- Container working directory -containerWorkingDir: "/" - -rbac: - # -- Enables Role/RoleBinding creation - create: true - - # -- Role/RoleBinding annotations - annotations: {} - - # -- Role/RoleBinding labels - extraLabels: {} - - # -- If true and `rbac.enabled`, will deploy a Role/RoleBinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - - # -- additional rules for a role - extraRules: [] - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - # -- Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dmnd-sre-vmagnt-prd-mds@meesho-demand-prd-0622.iam.gserviceaccount.com - } - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # -- mount API token to pod directly - automountToken: true - -# -- See `kubectl explain poddisruptionbudget.spec` for more or check [official documentation](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# -- Generates `remoteWrite.*` flags and config maps with value content for values, that are of type list of map. -# Each item should contain `url` param to pass validation. -remoteWrite: - - url: http://vm-insert-demand-prd.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write -# urlRelabelConfig: -# - action: keep -# source_labels: [env] -# regex: "dev" - -# -- VMAgent extra command line arguments -extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8429 - promscrape.config.strictParse: false # default is true - promscrape.maxScrapeSize: 1000000000 # default is 167772160 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - # promscrape.suppressScrapeErrors: true # default is false - - # Uncomment and specify the port if you want to support any of the protocols: - # https://docs.victoriametrics.com/victoriametrics/vmagent/#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for more details. -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth-secret - # key: password - -# -- Specify alternative source for env variables -envFrom: - [] - #- configMapRef: - # name: special-config - -# -- Extra labels for Deployment and Statefulset -extraLabels: - bu: "demand" - team: "demand-sre" - service: "vm-agent-demand-prd-secondary" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra labels for Pods only -podLabels: - bu: "demand" - team: "demand-sre" - service: "vm-agent-demand-prd-secondary" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra selector labels common for pod and service -selectorLabels: {} - -# -- Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# -- Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# -- Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -# -- Extra containers to run in a pod with vmagent -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -# -- Init containers for vmagent -initContainers: - [] - # - name: example - # image: example-image - -# -- Security context to be added to pod -podSecurityContext: - enabled: true - # fsGroup: 2000 - -# -- Security context to be added to pod's containers -securityContext: - enabled: true - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - # -- Enable agent service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - extraLabels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) for details - externalIPs: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8429 - # -- Target port - targetPort: http - # nodePort: 30000 - # -- Service type - type: ClusterIP - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - -ingress: - # -- Enable deployment of ingress for agent - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-agent-demand-prd-secondary.meeshogcp.in - path: - - / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - - # -- Ingress controller class name - ingressClassName: "nginx-internal" - - # -- Ingress path type - pathType: Prefix - -# -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) -resources: - requests: - cpu: 370 - memory: 650Gi - -# -- Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) -nodeSelector: - dedicated: "vmagent-c4d" - -# -- Node tolerations for server scheduling to nodes with taints. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-c4d" - effect: "NoSchedule" - -# -- Pod topologySpreadConstraints -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# -- Pod affinity -affinity: {} - -# -- VMAgent [scraping configuration](https://docs.victoriametrics.com/victoriametrics/vmagent/#how-to-collect-metrics-in-prometheus-format) -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-demand-prd-secondary" - -# -- Priority class to be assigned to the pod(s) -priorityClassName: "" - -# -- Enable the host network -hostNetwork: false - -serviceMonitor: - # -- Enable deployment of Service Monitor for server component. This is Prometheus operator object - enabled: false - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Service Monitor relabelings - relabelings: [] - # -- Basic auth params for Service Monitor - basicAuth: {} - # -- Service Monitor metricRelabelings - metricRelabelings: [] - # -- Service Monitor targetPort - targetPort: http - # interval: 15s - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - -# -- Empty dir configuration for a case, when persistence is disabled -emptyDir: {} - -persistentVolume: - # -- Create/use Persistent Volume Claim for server component. Empty dir if false - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Size of the volume. Should be calculated based on the logs you send and retention policy you set. - size: 10Gi - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume additional labels - extraLabels: {} - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Horizontal Pod Autoscaling. -# Note that it is not intended to be used for vmagents which perform scraping. -# In order to scale scraping vmagents check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) -horizontalPodAutoscaling: - # -- Use HPA for vmagent - enabled: false - # -- Maximum replicas for HPA to use to to scale vmagent - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale vmagent - minReplicas: 1 - # -- Metric for HPA to use to scale vmagent - metrics: [] - -# -- VMAgent scrape configuration -config: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -probe: - # -- Readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - # -- Liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - # -- Startup probe - startup: {} - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -allowedMetricsEndpoints: - - /metrics - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-demand-prd-ase1/victoriametrics-agent-shared/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/victoriametrics-agent-shared/custom-values.yaml deleted file mode 100644 index cffa5af..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/victoriametrics-agent-shared/custom-values.yaml +++ /dev/null @@ -1,481 +0,0 @@ -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "demand-shared-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -# -- Replica count -replicaCount: 2 - -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - -# -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables -deployment: - spec: - minReadySeconds: 1200 - progressDeadlineSeconds: 1800 - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - -image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - # -- Image tag, set to `Chart.AppVersion` by default - tag: v1.133.0 # rewrites Chart.AppVersion - # -- Variant of the image to use. - # e.g. enterprise, scratch - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - -# -- Image pull secrets -imagePullSecrets: [] - -# -- Add additional DNS entries to pods hosts file. Check [official documentation](https://kubernetes.io/docs/tasks/network/customize-hosts-file-for-pods/) -hostAliases: [] -# - ip: 192.168.1.1 -# hostNames: -# - test.example.com -# - another.example.net - -# -- Override chart name -nameOverride: "" - -# -- Override resources fullname -fullnameOverride: "vm-agent-demand-prd-shared" - -# -- Container working directory -containerWorkingDir: "/" - -rbac: - # -- Enables Role/RoleBinding creation - create: true - - # -- Role/RoleBinding annotations - annotations: {} - - # -- Role/RoleBinding labels - extraLabels: {} - - # -- If true and `rbac.enabled`, will deploy a Role/RoleBinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - - # -- additional rules for a role - extraRules: [] - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - # -- Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dmnd-sre-vmagnt-prd-mds@meesho-demand-prd-0622.iam.gserviceaccount.com - } - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # -- mount API token to pod directly - automountToken: true - -# -- See `kubectl explain poddisruptionbudget.spec` for more or check [official documentation](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# -- Generates `remoteWrite.*` flags and config maps with value content for values, that are of type list of map. -# Each item should contain `url` param to pass validation. -remoteWrite: - - url: http://vminsert-demand-prd.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write -# urlRelabelConfig: -# - action: keep -# source_labels: [env] -# regex: "dev" - -# -- VMAgent extra command line arguments -extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8429 - promscrape.config.strictParse: false # default is true - promscrape.maxScrapeSize: 1000000000 # default is 167772160 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - # promscrape.suppressScrapeErrors: true # default is false - - # Uncomment and specify the port if you want to support any of the protocols: - # https://docs.victoriametrics.com/victoriametrics/vmagent/#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for more details. -# GOGC: trade memory for lower CPU (GC) usage. Default 30; 100 reduces GC frequency. See https://github.com/VictoriaMetrics/VictoriaMetrics/issues/7832 -env: - - name: GOGC - value: "200" - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth-secret - # key: password - -# -- Specify alternative source for env variables -envFrom: - [] - #- configMapRef: - # name: special-config - -# -- Extra labels for Deployment and Statefulset -extraLabels: - bu: "demand" - team: "demand-sre" - service: "vm-agent-demand-prd-shared" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra labels for Pods only -podLabels: - bu: "demand" - team: "demand-sre" - service: "vm-agent-demand-prd-shared" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra selector labels common for pod and service -selectorLabels: {} - -# -- Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# -- Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# -- Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -# -- Extra containers to run in a pod with vmagent -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -# -- Init containers for vmagent -initContainers: - [] - # - name: example - # image: example-image - -# -- Security context to be added to pod -podSecurityContext: - enabled: true - # fsGroup: 2000 - -# -- Security context to be added to pod's containers -securityContext: - enabled: true - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - # -- Enable agent service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - extraLabels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) for details - externalIPs: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8429 - # -- Target port - targetPort: http - # nodePort: 30000 - # -- Service type - type: ClusterIP - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - -ingress: - # -- Enable deployment of ingress for agent - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-agent-demand-prd-shared.meeshogcp.in - path: - - / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - - # -- Ingress controller class name - ingressClassName: "nginx-internal" - - # -- Ingress path type - pathType: Prefix - -# -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) -resources: - requests: - cpu: 370 - memory: 650Gi - -# -- Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) -nodeSelector: - dedicated: "vmagent-c4d" - -# -- Node tolerations for server scheduling to nodes with taints. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-c4d" - effect: "NoSchedule" - -# -- Pod topologySpreadConstraints -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# -- Pod affinity -affinity: {} - -# -- VMAgent [scraping configuration](https://docs.victoriametrics.com/victoriametrics/vmagent/#how-to-collect-metrics-in-prometheus-format) -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-demand-prd-shared" - -# -- Priority class to be assigned to the pod(s) -priorityClassName: "" - -# -- Enable the host network -hostNetwork: false - -serviceMonitor: - # -- Enable deployment of Service Monitor for server component. This is Prometheus operator object - enabled: false - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Service Monitor relabelings - relabelings: [] - # -- Basic auth params for Service Monitor - basicAuth: {} - # -- Service Monitor metricRelabelings - metricRelabelings: [] - # -- Service Monitor targetPort - targetPort: http - # interval: 15s - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - -# -- Empty dir configuration for a case, when persistence is disabled -emptyDir: {} - -persistentVolume: - # -- Create/use Persistent Volume Claim for server component. Empty dir if false - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Size of the volume. Should be calculated based on the logs you send and retention policy you set. - size: 10Gi - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume additional labels - extraLabels: {} - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Horizontal Pod Autoscaling. -# Note that it is not intended to be used for vmagents which perform scraping. -# In order to scale scraping vmagents check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) -horizontalPodAutoscaling: - # -- Use HPA for vmagent - enabled: false - # -- Maximum replicas for HPA to use to to scale vmagent - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale vmagent - minReplicas: 1 - # -- Metric for HPA to use to scale vmagent - metrics: [] - -# -- VMAgent scrape configuration -config: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -probe: - # -- Readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - # -- Liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - # -- Startup probe - startup: {} - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -allowedMetricsEndpoints: - - /metrics - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-demand-prd-ase1/victoriametrics-agent/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index 8c7300e..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,485 +0,0 @@ -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "demand-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -# -- Replica count -replicaCount: 2 - -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - -# -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables -deployment: - spec: - minReadySeconds: 1200 - progressDeadlineSeconds: 1800 - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - -image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - # -- Image tag, set to `Chart.AppVersion` by default - tag: v1.133.0 # rewrites Chart.AppVersion - # -- Variant of the image to use. - # e.g. enterprise, scratch - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - -# -- Image pull secrets -imagePullSecrets: [] - -# -- Add additional DNS entries to pods hosts file. Check [official documentation](https://kubernetes.io/docs/tasks/network/customize-hosts-file-for-pods/) -hostAliases: [] -# - ip: 192.168.1.1 -# hostNames: -# - test.example.com -# - another.example.net - -# -- Override chart name -nameOverride: "" - -# -- Override resources fullname -fullnameOverride: "vm-agent-demand-prd" - -# -- Container working directory -containerWorkingDir: "/" - -rbac: - # -- Enables Role/RoleBinding creation - create: true - - # -- Role/RoleBinding annotations - annotations: {} - - # -- Role/RoleBinding labels - extraLabels: {} - - # -- If true and `rbac.enabled`, will deploy a Role/RoleBinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - - # -- additional rules for a role - extraRules: [] - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - # -- Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dmnd-sre-vmagnt-prd-mds@meesho-demand-prd-0622.iam.gserviceaccount.com - } - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # -- mount API token to pod directly - automountToken: true - -# -- See `kubectl explain poddisruptionbudget.spec` for more or check [official documentation](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# -- Generates `remoteWrite.*` flags and config maps with value content for values, that are of type list of map. -# Each item should contain `url` param to pass validation. -remoteWrite: - - url: http://vm-insert-demand-prd.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write - disableOnDiskQueue: false - dropSamplesOnOverload: false - - url: http://prd-census-server-demand.prd-census-server-demand.svc.cluster.local/api/v1/write - disableOnDiskQueue: true - dropSamplesOnOverload: true -# urlRelabelConfig: -# - action: keep -# source_labels: [env] -# regex: "dev" - -# -- VMAgent extra command line arguments -extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8429 - promscrape.config.strictParse: false # default is true - promscrape.maxScrapeSize: 1000000000 # default is 167772160 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - # promscrape.suppressScrapeErrors: true # default is false - - # Uncomment and specify the port if you want to support any of the protocols: - # https://docs.victoriametrics.com/victoriametrics/vmagent/#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for more details. -env: - - name: GOGC - value: "200" - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth-secret - # key: password - -# -- Specify alternative source for env variables -envFrom: - [] - #- configMapRef: - # name: special-config - -# -- Extra labels for Deployment and Statefulset -extraLabels: - bu: "demand" - team: "demand-sre" - service: "vm-agent-demand-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra labels for Pods only -podLabels: - bu: "demand" - team: "demand-sre" - service: "vm-agent-demand-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra selector labels common for pod and service -selectorLabels: {} - -# -- Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# -- Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# -- Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -# -- Extra containers to run in a pod with vmagent -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -# -- Init containers for vmagent -initContainers: - [] - # - name: example - # image: example-image - -# -- Security context to be added to pod -podSecurityContext: - enabled: true - # fsGroup: 2000 - -# -- Security context to be added to pod's containers -securityContext: - enabled: true - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - # -- Enable agent service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - extraLabels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) for details - externalIPs: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8429 - # -- Target port - targetPort: http - # nodePort: 30000 - # -- Service type - type: ClusterIP - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - -ingress: - # -- Enable deployment of ingress for agent - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-agent-demand-prd.meeshogcp.in - path: - - / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - - # -- Ingress controller class name - ingressClassName: "nginx-internal" - - # -- Ingress path type - pathType: Prefix - -# -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) -resources: - requests: - cpu: 130 - memory: 250Gi - -# -- Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) -nodeSelector: - dedicated: "vmagent-c4d" - -# -- Node tolerations for server scheduling to nodes with taints. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-c4d" - effect: "NoSchedule" - -# -- Pod topologySpreadConstraints -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# -- Pod affinity -affinity: {} - -# -- VMAgent [scraping configuration](https://docs.victoriametrics.com/victoriametrics/vmagent/#how-to-collect-metrics-in-prometheus-format) -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-demand-prd" - -# -- Priority class to be assigned to the pod(s) -priorityClassName: "" - -# -- Enable the host network -hostNetwork: false - -serviceMonitor: - # -- Enable deployment of Service Monitor for server component. This is Prometheus operator object - enabled: false - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Service Monitor relabelings - relabelings: [] - # -- Basic auth params for Service Monitor - basicAuth: {} - # -- Service Monitor metricRelabelings - metricRelabelings: [] - # -- Service Monitor targetPort - targetPort: http - # interval: 15s - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - -# -- Empty dir configuration for a case, when persistence is disabled -emptyDir: {} - -persistentVolume: - # -- Create/use Persistent Volume Claim for server component. Empty dir if false - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Size of the volume. Should be calculated based on the logs you send and retention policy you set. - size: 10Gi - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume additional labels - extraLabels: {} - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Horizontal Pod Autoscaling. -# Note that it is not intended to be used for vmagents which perform scraping. -# In order to scale scraping vmagents check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) -horizontalPodAutoscaling: - # -- Use HPA for vmagent - enabled: false - # -- Maximum replicas for HPA to use to to scale vmagent - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale vmagent - minReplicas: 1 - # -- Metric for HPA to use to scale vmagent - metrics: [] - -# -- VMAgent scrape configuration -config: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -probe: - # -- Readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - # -- Liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - # -- Startup probe - startup: {} - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -allowedMetricsEndpoints: - - /metrics - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-demand-prd-ase1/victoriametrics-insert/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/victoriametrics-insert/custom-values.yaml deleted file mode 100644 index 1c13675..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/victoriametrics-insert/custom-values.yaml +++ /dev/null @@ -1,411 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: false - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- IDs of vmstorage nodes to exclude from writing - excludeStorageIDs: [] - # -- Override default `app` label name - name: vminsert - # -- VMInsert strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - # override Chart.AppVersion - tag: v1.107.0-cluster - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMInsert http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-demand-prd - # -- Extra command line arguments for vminsert component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8480 - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-demand-prd-0.vm-storage-demand-prd.victoriametrics.svc:8400" - - "vm-storage-demand-prd-1.vm-storage-demand-prd.victoriametrics.svc:8400" - - "vm-storage-demand-prd-2.vm-storage-demand-prd.victoriametrics.svc:8400" - - "vm-storage-demand-prd-3.vm-storage-demand-prd.victoriametrics.svc:8400" - - "vm-storage-demand-prd-4.vm-storage-demand-prd.victoriametrics.svc:8400" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "demand" - team: "demand-sre" - service: "vm-insert-demand-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Pod's additional labels - podLabels: - bu: "demand" - team: "demand-sre" - service: "vm-insert-demand-prd" - env: "prd" - priority: "p0" - type: "vminsert" - terminationGracePeriodSeconds: 30 - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - - # -- Readiness & Liveness probes - probe: - # -- VMInsert readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert startup probe - startup: {} - - # -- Relabel configuration - relabel: - enabled: false - config: [] - # -- Use existing configmap if specified - # otherwise .config values will be used. Relabel config **should** reside under `relabel.yml` key - configMap: "" - - # Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 7 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vminsert - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vminsert - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - dedicated: "vminsert-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 20 - memory: 10Gi - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - service: - # -- Create VMInsert service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here]( https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) for details - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-insert-demand-prd.meeshogcp.in - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - - # -- Ingress controller class name - ingressClassName: nginx-internal - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for insert component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/insert" .Values.vminsert }}' - - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vminsert component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-demand-prd-ase1/victoriametrics-select/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/victoriametrics-select/custom-values.yaml deleted file mode 100644 index c9aedfb..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/victoriametrics-select/custom-values.yaml +++ /dev/null @@ -1,449 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - splitService: true - externalService: - enabled: true - name: vmselect-demand-prd-proxy - - # -- Enable clusternative service for LoadBalancer vmselect component - clusternativeService: - enabled: false - # -- Override default `app` label name - name: "" - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # override Chart.AppVersion - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMSelect http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-demand-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - # -- Extra command line arguments for vmselect component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8481 - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-demand-prd-0.vm-storage-demand-prd.victoriametrics.svc:8401" - - "vm-storage-demand-prd-1.vm-storage-demand-prd.victoriametrics.svc:8401" - - "vm-storage-demand-prd-2.vm-storage-demand-prd.victoriametrics.svc:8401" - - "vm-storage-demand-prd-3.vm-storage-demand-prd.victoriametrics.svc:8401" - - "vm-storage-demand-prd-4.vm-storage-demand-prd.victoriametrics.svc:8401" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "demand" - team: "demand-sre" - service: "vm-select-demand-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # -- Pod's additional labels - podLabels: - bu: "demand" - team: "demand-sre" - service: "vm-select-demand-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Readiness & Liveness probes - probe: - # -- VMSelect readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 20 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 7 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vmselect - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vmselect - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - dedicated: "vmselect-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 36 - memory: 60Gi - - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Create VMSelect service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-select-demand.prd.meesho.int - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # -- Ingress controller class name - ingressClassName: contour-internal-0 - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for select component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/select" .Values.vmselect }}' - - # -- vmselect mode: deployment, daemonSet - mode: deployment - - # -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables - deployment: - spec: - # -- VMSelect strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - - # -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables - statefulSet: - enabled: false - spec: - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume extra labels - extraLabels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmselect component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: false - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - diff --git a/helm-overrides/k8s-demand-prd-ase1/victoriametrics-storage/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1/victoriametrics-storage/custom-values.yaml deleted file mode 100644 index 1b38b34..0000000 --- a/helm-overrides/k8s-demand-prd-ase1/victoriametrics-storage/custom-values.yaml +++ /dev/null @@ -1,363 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - enabled: false - -vminsert: - enabled: false - -vmauth: - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- Override default `app` label name - name: vmstorage - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Variant of the image to use. e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMStorage http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-demand-prd - - # -- Additional environment variables (ex.: secret tokens, flags). Check https://docs.victoriametrics.com/victoriametrics/#environment-variables for details - env: [] - # -- Specify alternative source for env variables - envFrom: [] - # -- Data retention period. Possible units character: h(ours), d(ays), w(eeks), y(ears), if no unit character specified - month. The minimum retention period is 24h. - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - httpListenAddr: :8482 - loggerTimezone: "Asia/Kolkata" - dedup.minScrapeInterval: 60s - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage-n4d" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage-n4d" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - # -- Use an alternate scheduler, e.g. "stork". Check https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ for details - # schedulerName: - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Override Persistent Volume Claim name - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are https://kubernetes.io/docs/concepts/storage/persistent-volumes/ - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume extra labels - extraLabels: {} - # -- Storage class name. Will be empty if not set - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume - size: 9677Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "demand" - team: "demand-sre" - service: "vm-storage-demand-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - # -- Pod's additional labels - podLabels: - bu: "demand" - team: "demand-sre" - service: "vm-storage-demand-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Details are https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 85 - memory: 700Gi - - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: - enabled: false - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - podSecurityContext: - enabled: false - service: - enabled: true - # -- Service annotations - annotations: {} - # -- Service ClusterIP - clusterIP: None - # -- Service type - type: ClusterIP - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmselect - vmselectPort: 8401 - # -- Extra service ports - extraPorts: [] - # -- Health check node port for a service. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilyPolicy: "" - # -- List of service IP families. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilies: [] - # -- Traffic Distribution. Check https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution - trafficDistribution: "" - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - minReadySeconds: 5 - # -- Readiness probes - probe: - # -- VMStorage readiness probe - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMStorage startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmstorage component - enabled: false - # -- Maximum replicas for HPA to use to to scale the vmstorage component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmstorage component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmstorage component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: - scaleDown: - selectPolicy: Disabled - - vmbackupmanager: - # -- Enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-demand-prd-ase1c/conntrack-adjuster/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 7b64b15..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - # Maximum number of conntrack entries - max: 2097152 - # Hash size for conntrack - hashsize: 524288 - # Sleep interval between adjustments (seconds) - sleepInterval: 30 - -# Enable additional matchExpressions -# addExtraMatchExpressions: true - -# Additional matchExpressions to append -# additionalMatchExpressions: -# - key: node-role -# operator: In -# values: -# - "worker" -# - "ingress" -# - key: environment -# operator: In -# values: -# - "production" -# - "staging" \ No newline at end of file diff --git a/helm-overrides/k8s-demand-prd-ase1c/contour-cert-checker/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/contour-cert-checker/custom-values.yaml deleted file mode 100644 index 6d1f4be..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=k8s-demand-prd-ase1c"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - dedicated: demand-devops - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: demand-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/k8s-demand-prd-ase1c/contour-external/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/contour-external/custom-values.yaml deleted file mode 100644 index 107eecf..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/contour-external/custom-values.yaml +++ /dev/null @@ -1,85 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - network: - num-trusted-hops: 1 -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: demand - team: demand-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external - nodeSelector: - dedicated: contour-external - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: demand - team: demand-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external - nodeSelector: - dedicated: contour-external - logLevel: error - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 2500m - memory: 512Mi - service: - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-demand-c-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-demand-prd-ase1c/contour-internal-0/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/contour-internal-0/custom-values.yaml deleted file mode 100644 index faf3797..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,97 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: demand - team: demand-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 4Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - updateStrategy: - rollingUpdate: - maxSurge: 10% - enabled: true - podLabels: - bu: demand - team: demand-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 350 - targetCPU: "55" - targetMemory: "55" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 12 - memory: 8Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-demand-c-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-demand-prd-ase1c/contour-internal-1/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/contour-internal-1/custom-values.yaml deleted file mode 100644 index becd4cd..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,89 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: demand - team: demand-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 8 - memory: 11Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1-new - nodeSelector: - dedicated: contour-internal-1-new - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: demand - team: demand-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - logLevel: error - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 350 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int1-demand-c-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-demand-prd-ase1c/contour-internal-intra-0/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index 5339484..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,95 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: demand - team: demand-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 4Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - updateStrategy: - rollingUpdate: - maxSurge: 10% - enabled: true - podLabels: - bu: demand - team: demand-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 350 - targetCPU: "55" - targetMemory: "55" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 12 - memory: 8Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-demand-prd-ase1c/contour-internal-intra-1/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index 6a8efc7..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,87 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 7 - podLabels: - bu: demand - team: demand-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 4Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1-new - nodeSelector: - dedicated: contour-internal-1-new - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: demand - team: demand-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 350 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-demand-prd-ase1c/coredns/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/coredns/custom-values.yaml deleted file mode 100644 index db0d1c8..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/coredns/custom-values.yaml +++ /dev/null @@ -1,30 +0,0 @@ -replicaCount: 16 - -labels: - bu: demand - team: demand-devops - env: prd - -clusterIP: 10.209.16.2 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "demand-devops" - effect: "NoSchedule" - -nodeSelector: - dedicated: demand-devops - kubernetes.io/os: linux - -overwriteRegion: "c" -communicationType: "" \ No newline at end of file diff --git a/helm-overrides/k8s-demand-prd-ase1c/external-secrets/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/external-secrets/custom-values.yaml deleted file mode 100644 index 2867e35..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/external-secrets/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Additional service annotations - annotations: {} - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: demand-devops - nodeSelector: - dedicated: demand-devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: demand-devops - nodeSelector: - dedicated: demand-devops - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: demand-devops - nodeSelector: - dedicated: demand-devops diff --git a/helm-overrides/k8s-demand-prd-ase1c/flagger/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/flagger/custom-values.yaml deleted file mode 100644 index 54ce43e..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/flagger/custom-values.yaml +++ /dev/null @@ -1,69 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - dedicated: demand-devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: demand-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: demand-devops - bu: infra - region: ase1c - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-demand-rollout-service.prd-demand-rollout-service.svc.cluster.local" - - -rolloutService: - enabled: true - apps: - prd-devops-node-app-consumers: true - prd-devops-node-app: true - prd-devops-golang-app: true - prd-devops-golang-app-test: true - prd-devops-golang-app-crons: true diff --git a/helm-overrides/k8s-demand-prd-ase1c/fluentd/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/fluentd/custom-values.yaml deleted file mode 100644 index 572b8f2..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/fluentd/custom-values.yaml +++ /dev/null @@ -1,720 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-dmnd-dmsre-fluentd-prd@meesho-demand-ase1c-prd-0625.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: demand - team: sre - type: fluentd - service: fluentd-demand-prd - priority: p0 - env: prd - zone_extended: ase1c - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: demand - team: sre - type: fluentd - service: fluentd-demand-prd - priority: p0 - env: prd - zone_extended: ase1c - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: [] -# - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - # - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-demand-prd-ase1c/ingress-nginx/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/ingress-nginx/custom-values.yaml deleted file mode 100644 index 4caac2e..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/ingress-nginx/custom-values.yaml +++ /dev/null @@ -1,32 +0,0 @@ -ingress-nginx: - controller: - config: - proxy-body-size: "50g" - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - resources: - requests: - cpu: 200m - memory: 512Mi - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 30 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-demand-ase1c-prd"}}}' - nodeSelector: - dedicated: devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-demand-prd-ase1c/keda/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/keda/custom-values.yaml deleted file mode 100644 index b32245f..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/keda/custom-values.yaml +++ /dev/null @@ -1,35 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: demand-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "demand-devops" - effect: "NoSchedule" - podLabels: - bu: "demand" - team: "demand-devops" - metricsAdapter: - bu: "demand" - team: "demand-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 100m - memory: 200Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 50m - memory: 150Mi - requests: - cpu: 10m - memory: 80Mi \ No newline at end of file diff --git a/helm-overrides/k8s-demand-prd-ase1c/kube-dns/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/kube-dns/custom-values.yaml deleted file mode 100644 index 4f4e42b..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.209.16.2"],"prd.mrouter.int.svc.cluster.local":["10.209.16.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.209.16.2"]} diff --git a/helm-overrides/k8s-demand-prd-ase1c/kube-events/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/kube-events/custom-values.yaml deleted file mode 100644 index 4bd0386..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-demand-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - dedicated: demand-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: demand-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: demand-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: demand-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: demand-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: demand-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-demand-prd-ase1c/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 745047c..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-demand-ase1c-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-demand-ase1c-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "demand" - team: "demand-sre" - service: "kube-state-metrics-demand-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "demand-devops" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "demand-devops" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 23m - memory: 169Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-demand-prd-ase1c/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index 88dff9a..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,119 +0,0 @@ -fullnameOverride: opentelemetry-demand-ase1c-prd - -mode: daemonset - -priorityClassName: "system-node-critical" - -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-central-prd-ase1c-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external - - key: node_pool - operator: NotIn - values: - - np-demand-default-prd-ase1 - -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 100m - memory: 124Mi - limits: - cpu: 200m - memory: 256Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "demand" - team: "sre" - service: "opentelemetry-demand-ase1c-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - zone_extended: "ase1c" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/k8s-demand-prd-ase1c/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index c9158b8..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,498 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-admin-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - - -# Extra labels to be added to node exporter pods -podLabels: - bu: "demand" - team: "demand-sre" - service: "node-exporter-demand-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - zone_extended: "ase1c" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-demand-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index 6ed7d12..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-demand-ase1c-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-demand-ase1c-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "demand" - team: "demand-sre" - service: "stackdriver-exporter-demand-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - zone_extended: "ase1c" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-demand-ase1c-prd-0225" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,compute.googleapis.com/instance,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - dedicated: "vmselect" - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-cntr-cnsre-stackdriver-prd@meesho-demand-ase1c-prd-0622.iam.gserviceaccount.com - } - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/k8s-demand-prd-ase1c/telegraf-operator/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/telegraf-operator/custom-values.yaml deleted file mode 100644 index 075799c..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,151 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "demand" - team: "demand-sre" - service: "telegraf-operator-demand-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - zone_extended: "ase1c" - -nodeSelector: - dedicated: "vmselect" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false diff --git a/helm-overrides/k8s-demand-prd-ase1c/victoria-metrics-agent/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index 82bf980..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,298 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-demand-ase1c-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-demand-ase1c-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-common-np-dmnd-prd@meesho-demand-ase1c-prd-0625.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - http://vminsert-demand-ase1c-prd.cntr-c.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-demand-ase1c-prd.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write - # - http://vminsert-demand-ase1c-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "demand" - team: "demand-sre" - service: "vmagent-demand-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmagent" - zone_extended: "ase1c" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-demand-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-demand-ase1c-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 19 - memory: 19Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-demand-prd-ase1c/victoria-metrics-insert/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index 6078b1d..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,235 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-demand-ase1c-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-demand-ase1c-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "demand" - team: "demand-sre" - service: "vminsert-demand-ase1c-prd" - env: "prd" - priority: "p0" - type: "vminsert" - zone_extended: "ase1c" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 1 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vminsert" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 3 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 4 - memory: 2Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-demand-ase1c-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-demand-prd-ase1c/victoria-metrics-select/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index 62f5a8d..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,301 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-demand-ase1c-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-demand-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - clusternative.maxConcurrentRequests: "384" - dedup.minScrapeInterval: 60s - search.maxConcurrentRequests: "384" - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "demand" - team: "demand-sre" - service: "vmselect-demand-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmselect" - zone_extended: "ase1c" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 15 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 1 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "vmselect" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 18 - memory: 25Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-demand-ase1c-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - clusternativeService: - enabled: true - port: 8401 - sessionAffinity: None - ipFamilyPolicy: SingleStack - - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-demand-prd-ase1c/victoria-metrics-storage/custom-values.yaml b/helm-overrides/k8s-demand-prd-ase1c/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index 6b3f170..0000000 --- a/helm-overrides/k8s-demand-prd-ase1c/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,317 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-demand-ase1c-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 500Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "demand" - team: "demand-sre" - service: "vmstorage-demand-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - zone_extended: "ase1c" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 16 - memory: 275Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.93.7-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-demand-ase1c-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/README.md b/helm-overrides/k8s-dengspark-di-prd-ase1/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/conntrack-adjuster/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 7b64b15..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - # Maximum number of conntrack entries - max: 2097152 - # Hash size for conntrack - hashsize: 524288 - # Sleep interval between adjustments (seconds) - sleepInterval: 30 - -# Enable additional matchExpressions -# addExtraMatchExpressions: true - -# Additional matchExpressions to append -# additionalMatchExpressions: -# - key: node-role -# operator: In -# values: -# - "worker" -# - "ingress" -# - key: environment -# operator: In -# values: -# - "production" -# - "staging" \ No newline at end of file diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/external-secrets/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/external-secrets/custom-values.yaml deleted file mode 100644 index 31ca194..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/external-secrets/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Additional service annotations - annotations: {} - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dengspark-di-devops - nodeSelector: - dedicated: dengspark-di-devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dengspark-di-devops - nodeSelector: - dedicated: dengspark-di-devops - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dengspark-di-devops - nodeSelector: - dedicated: dengspark-di-devops diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/flagger/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/flagger/custom-values.yaml deleted file mode 100644 index 2826f3b..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/flagger/custom-values.yaml +++ /dev/null @@ -1,55 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: hpa-changes-26 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - dedicated: dengspark-di-devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dengspark-di-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: devops - bu: dengspark diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/fluentd/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/fluentd/custom-values.yaml deleted file mode 100644 index 3e9307d..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/fluentd/custom-values.yaml +++ /dev/null @@ -1,764 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-dpspsre-fluentd-prd@meesho-dengspark-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: dengspark - team: sre - type: fluentd - service: fluentd-dengspark-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: dengspark - team: sre - type: fluentd - service: fluentd-dengspark-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - - -# externalSecret: -# enabled: false -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: [] -# - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/ingress-nginx-external/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/ingress-nginx-external/custom-values.yaml deleted file mode 100644 index beb2995..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/ingress-nginx-external/custom-values.yaml +++ /dev/null @@ -1,30 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 15 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - ingressClass: nginx-external - ingressClassByName: true - watchIngressWithoutClass: false - ingressClassResource: - controllerValue: k8s.io/ingress-nginx-external - name: nginx-external - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dengspark-di-external-prd"}}}' - nodeSelector: - dedicated: dengspark-di-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "dengspark-di-devops" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/ingress-nginx-internal/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/ingress-nginx-internal/custom-values.yaml deleted file mode 100644 index 3e90a92..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/ingress-nginx-internal/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 5 - maxReplicas: 15 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dpsp-di-int-prd"}}}' - nodeSelector: - dedicated: dengspark-di-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "dengspark-di-devops" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/keda/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/keda/custom-values.yaml deleted file mode 100644 index 875bf49..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/keda/custom-values.yaml +++ /dev/null @@ -1,45 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: dengspark-di-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "dengspark-di-devops" - effect: "NoSchedule" - podLabels: - bu: "dengspark" - team: "dengspark-devops" - metricsAdapter: - bu: "dengspark" - team: "dengspark-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 2000Mi - requests: - cpu: 400m - memory: 500Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 30m - memory: 100Mi - metricServer: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 200m - memory: 200Mi - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/kube-events/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/kube-events/custom-values.yaml deleted file mode 100644 index 91a531c..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-dengspark-di-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - dedicated: dengspark-di-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: dengspark-di-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: dengspark-di-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: dengspark-di-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: dengspark-di-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: dengspark-di-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 782d20c..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-dengspark-di-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-dengspark-di-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "kube-state-metrics-dengspark-di-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "victoriametrics" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 10m - memory: 50Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index 2caf400..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,496 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 200m - # memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-admin-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "node-exporter-dengspark-di-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-agent/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index 89021d8..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,295 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-dengspark-di-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-dengspark-di-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dsprk-spsre-vmagent-prd@meesho-dengspark-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-dengspark.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-dengspark-di-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vmagent-dengspark-di-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dengspark-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dengspark-di-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 2 - memory: 4Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "victoriametrics" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-alert/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-alert/custom-values.yaml deleted file mode 100644 index b696ecd..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-alert/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-dengspark-di-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-dengspark-di-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-dengspark-di-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-dengspark-di-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.meeshogcp.in" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/dengspark/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-dengspark-di-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-dengspark-di-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - dedicated: "vminsert" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-insert/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index a1db3f3..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,234 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dengspark-di-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-dengspark-di-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vminsert-dengspark-di-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 2 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vminsert" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 2 - memory: 2Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-dengspark-di-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-select/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index 9a085ee..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,293 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dengspark-di-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-dengspark-di-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 96 - clusternative.maxConcurrentRequests: 256 - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vmselect-dengspark-di-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 20 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "victoriametrics" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 1200m - memory: 4Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-dengspark-di-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-storage/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index ded1562..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,316 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-dengspark-di-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 30d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 200Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vmstorage-dengspark-di-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 4 - memory: 20Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.91.3-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-demand-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/victoriametrics-agent/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index fca4c95..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,319 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "dengspark-di-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -replicaCount: 2 -mode: deployment - -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-dengspark-di-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.133.0 # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "vm-agent-dengspark-di-prd" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dsprk-spsre-vmagent-prd@meesho-dengspark-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWrite: - - url: http://vm-insert-dengspark-di-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write - - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vm-agent-dengspark-di-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -podLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vm-agent-dengspark-di-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dengspark-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vm-agent-dengspark-di-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 4 - memory: 6Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent-n4d" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-n4d" - effect: "NoSchedule" - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/victoriametrics-insert/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/victoriametrics-insert/custom-values.yaml deleted file mode 100644 index 907c3e8..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/victoriametrics-insert/custom-values.yaml +++ /dev/null @@ -1,313 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - - - -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-mqkafka-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.107.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-dengspark-di-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-dengspark-di-prd-0.vm-storage-dengspark-di-prd.victoriametrics.svc:8400" - - "vm-storage-dengspark-di-prd-1.vm-storage-dengspark-di-prd.victoriametrics.svc:8400" - - "vm-storage-dengspark-di-prd-2.vm-storage-dengspark-di-prd.victoriametrics.svc:8400" - - "vm-storage-dengspark-di-prd-3.vm-storage-dengspark-di-prd.victoriametrics.svc:8400" - - "vm-storage-dengspark-di-prd-4.vm-storage-dengspark-di-prd.victoriametrics.svc:8400" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vm-insert-dengspark-di-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # arch: "arm64" - # runpod: "ondemand" - - podLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vm-insert-dengspark-di-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 2 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vm-common" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vm-common" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 2 - memory: 2Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-dengspark-di-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/victoriametrics-select/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/victoriametrics-select/custom-values.yaml deleted file mode 100644 index a39667c..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/victoriametrics-select/custom-values.yaml +++ /dev/null @@ -1,331 +0,0 @@ -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dengspark-di-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - clusternativeService: - enabled: false - - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-dengspark-di-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - externalService: - enabled: true - name: vmselect-dengspark-di-prd-proxy - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-dengspark-di-prd-0.vm-storage-dengspark-di-prd.victoriametrics.svc:8401" - - "vm-storage-dengspark-di-prd-1.vm-storage-dengspark-di-prd.victoriametrics.svc:8401" - - "vm-storage-dengspark-di-prd-2.vm-storage-dengspark-di-prd.victoriametrics.svc:8401" - - "vm-storage-dengspark-di-prd-3.vm-storage-dengspark-di-prd.victoriametrics.svc:8401" - - "vm-storage-dengspark-di-prd-4.vm-storage-dengspark-di-prd.victoriametrics.svc:8401" - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vm-select-dengspark-di-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - podLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vm-select-dengspark-di-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 20 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "vm-common" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vm-common" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 1200m - memory: 3Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vm-select-dengspark-di-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dengspark-di-prd-ase1/victoriametrics-storage/custom-values.yaml b/helm-overrides/k8s-dengspark-di-prd-ase1/victoriametrics-storage/custom-values.yaml deleted file mode 100644 index d6467eb..0000000 --- a/helm-overrides/k8s-dengspark-di-prd-ase1/victoriametrics-storage/custom-values.yaml +++ /dev/null @@ -1,346 +0,0 @@ -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: false - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-dengspark-di-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 30d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage-n4d" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage-n4d" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2000Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vm-storage-dengspark-di-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - podLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vm-storage-dengspark-di-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 2500m - memory: 25Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.91.3-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-demand-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dengspark-notebook-prd-ase1/README.md b/helm-overrides/k8s-dengspark-notebook-prd-ase1/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/k8s-dengspark-notebook-prd-ase1/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/k8s-dengspark-notebook-prd-ase1/external-secrets/custom-values.yaml b/helm-overrides/k8s-dengspark-notebook-prd-ase1/external-secrets/custom-values.yaml deleted file mode 100644 index fc2c304..0000000 --- a/helm-overrides/k8s-dengspark-notebook-prd-ase1/external-secrets/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Additional service annotations - annotations: {} - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dengspark-notebook-devops - nodeSelector: - dedicated: dengspark-notebook-devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dengspark-notebook-devops - nodeSelector: - dedicated: dengspark-notebook-devops - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dengspark-notebook-devops - nodeSelector: - dedicated: dengspark-notebook-devops diff --git a/helm-overrides/k8s-dengspark-notebook-prd-ase1/flagger/custom-values.yaml b/helm-overrides/k8s-dengspark-notebook-prd-ase1/flagger/custom-values.yaml deleted file mode 100644 index 08750d3..0000000 --- a/helm-overrides/k8s-dengspark-notebook-prd-ase1/flagger/custom-values.yaml +++ /dev/null @@ -1,55 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: hpa-changes-26 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - dedicated: dengspark-notebook-devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dengspark-notebook-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: devops - bu: dengspark diff --git a/helm-overrides/k8s-dengspark-notebook-prd-ase1/fluentd/custom-values.yaml b/helm-overrides/k8s-dengspark-notebook-prd-ase1/fluentd/custom-values.yaml deleted file mode 100644 index 2ef6fe1..0000000 --- a/helm-overrides/k8s-dengspark-notebook-prd-ase1/fluentd/custom-values.yaml +++ /dev/null @@ -1,733 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-dpspsre-fluentd-prd@meesho-dengspark-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: dengspark - team: sre - type: fluentd - service: fluentd-dengspark-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: dengspark - team: sre - type: fluentd - service: fluentd-dengspark-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - - -# externalSecret: -# enabled: false -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: - - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-dengspark-notebook-prd-ase1/ingress-nginx-external/custom-values.yaml b/helm-overrides/k8s-dengspark-notebook-prd-ase1/ingress-nginx-external/custom-values.yaml deleted file mode 100644 index 40f7841..0000000 --- a/helm-overrides/k8s-dengspark-notebook-prd-ase1/ingress-nginx-external/custom-values.yaml +++ /dev/null @@ -1,30 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 15 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - ingressClass: nginx-external - ingressClassByName: true - watchIngressWithoutClass: false - ingressClassResource: - controllerValue: k8s.io/ingress-nginx-external - name: nginx-external - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dengspark-notebook-external-prd"}}}' - nodeSelector: - dedicated: dengspark-notebook-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "dengspark-notebook-devops" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-dengspark-notebook-prd-ase1/ingress-nginx-internal/custom-values.yaml b/helm-overrides/k8s-dengspark-notebook-prd-ase1/ingress-nginx-internal/custom-values.yaml deleted file mode 100644 index 4ee323a..0000000 --- a/helm-overrides/k8s-dengspark-notebook-prd-ase1/ingress-nginx-internal/custom-values.yaml +++ /dev/null @@ -1,31 +0,0 @@ -ingress-nginx: - fullnameOverride: "nginx-int-dng-nb-prd-ingress-nginx" - controller: - config: - use-forwarded-headers: 'true' - enable-underscores-in-headers: 'true' - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 2 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dpsp-nb-int-prd"}}}' - allowSnippetAnnotations: true - nodeSelector: - dedicated: dengspark-notebook-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "dengspark-notebook-devops" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-dengspark-notebook-prd-ase1/keda/custom-values.yaml b/helm-overrides/k8s-dengspark-notebook-prd-ase1/keda/custom-values.yaml deleted file mode 100644 index 16a4b44..0000000 --- a/helm-overrides/k8s-dengspark-notebook-prd-ase1/keda/custom-values.yaml +++ /dev/null @@ -1,38 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: dengspark-notebook-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "dengspark-notebook-devops" - effect: "NoSchedule" - podLabels: - bu: "dengspark" - team: "dengspark-devops" - metricsAdapter: - bu: "dengspark" - team: "dengspark-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 2000Mi - requests: - cpu: 250m - memory: 300Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 30m - memory: 100Mi - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' \ No newline at end of file diff --git a/helm-overrides/k8s-dengspark-notebook-prd-ase1/kube-events/custom-values.yaml b/helm-overrides/k8s-dengspark-notebook-prd-ase1/kube-events/custom-values.yaml deleted file mode 100644 index 3b093d3..0000000 --- a/helm-overrides/k8s-dengspark-notebook-prd-ase1/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-dengspark-di-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - dedicated: dengspark-notebook-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: dengspark-notebook-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: dengspark-notebook-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: dengspark-notebook-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: dengspark-notebook-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: dengspark-notebook-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-dengspark-notebook-prd-ase1/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-dengspark-notebook-prd-ase1/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 751acd9..0000000 --- a/helm-overrides/k8s-dengspark-notebook-prd-ase1/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-dengspark-notebook-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-dengspark-notebook-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "kube-state-metrics-dengspark-notebook-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "victoriametrics" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 10m - memory: 50Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-dengspark-notebook-prd-ase1/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-dengspark-notebook-prd-ase1/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index 2caf400..0000000 --- a/helm-overrides/k8s-dengspark-notebook-prd-ase1/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,496 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 200m - # memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-admin-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "node-exporter-dengspark-di-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-dengspark-notebook-prd-ase1/rancher/custom-values.yaml b/helm-overrides/k8s-dengspark-notebook-prd-ase1/rancher/custom-values.yaml deleted file mode 100644 index 95cd7b7..0000000 --- a/helm-overrides/k8s-dengspark-notebook-prd-ase1/rancher/custom-values.yaml +++ /dev/null @@ -1,33 +0,0 @@ -rancher: - rancherImage: 'rancher/rancher' - rancherImageTag: latest - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dengspark-notebook-devops - nodeSelector: dengspark-notebook-devops - replicas: 2 - resources: - requests: - memory: 5G - cpu: 4 - limits: - memory: 5G - cpu: 4 - hostname: "rancher-dengspark-prd.meeshogcp.in" - serverURL: https://rancher-dengspark-prd.meeshogcp.in - ingress: - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: false - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: false - nginx.ingress.kubernetes.io/use-forwarded-headers: true - ingressClassName: nginx-internal - tls: - source: secret - secretName: tls-rancher-internal-ca - tls: external - postDelete: - enabled: false -bootstrapPassword: welcome@123 \ No newline at end of file diff --git a/helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-agent/custom-values.yaml b/helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index 18333d1..0000000 --- a/helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,295 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-dengspark-notebook-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-dengspark-notebook-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dsprk-spsre-vmagent-prd@meesho-dengspark-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-dengspark.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-dengspark-notebook-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vmagent-dengspark-notebook-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dengspark-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dengspark-notebook-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 2 - memory: 4Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "victoriametrics" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-alert/custom-values.yaml b/helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-alert/custom-values.yaml deleted file mode 100644 index 03c820c..0000000 --- a/helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-alert/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-dengspark-notebook-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-dengspark-notebook-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-dengspark-notebook-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-dengspark-notebook-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.meeshogcp.in" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/dengspark/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-dengspark-notebook-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-dengspark-notebook-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - dedicated: "vminsert" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-insert/custom-values.yaml b/helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index 620ba7f..0000000 --- a/helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,234 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dengspark-notebook-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-dengspark-notebook-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vminsert-dengspark-notebook-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 2 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vminsert" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 2 - memory: 2Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-dengspark-notebook-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-select/custom-values.yaml b/helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index 6ca9aa8..0000000 --- a/helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,293 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dengspark-notebook-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-dengspark-notebook-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 96 - clusternative.maxConcurrentRequests: 256 - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vmselect-dengspark-notebook-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 20 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "victoriametrics" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 1200m - memory: 4Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-dengspark-notebook-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-storage/custom-values.yaml b/helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index 9eab0cd..0000000 --- a/helm-overrides/k8s-dengspark-notebook-prd-ase1/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,316 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-dengspark-notebook-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "victoriametrics" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 200Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vmstorage-dengspark-notebook-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 2 - memory: 7Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.91.3-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-demand-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dengspark-prd-ase1/README.md b/helm-overrides/k8s-dengspark-prd-ase1/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/k8s-dengspark-prd-ase1/conntrack-adjuster/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 7b64b15..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - # Maximum number of conntrack entries - max: 2097152 - # Hash size for conntrack - hashsize: 524288 - # Sleep interval between adjustments (seconds) - sleepInterval: 30 - -# Enable additional matchExpressions -# addExtraMatchExpressions: true - -# Additional matchExpressions to append -# additionalMatchExpressions: -# - key: node-role -# operator: In -# values: -# - "worker" -# - "ingress" -# - key: environment -# operator: In -# values: -# - "production" -# - "staging" \ No newline at end of file diff --git a/helm-overrides/k8s-dengspark-prd-ase1/external-secrets/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/external-secrets/custom-values.yaml deleted file mode 100644 index fb5812b..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/external-secrets/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Additional service annotations - annotations: {} - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dengspark-devops - nodeSelector: - dedicated: dengspark-devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dengspark-devops - nodeSelector: - dedicated: dengspark-devops - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dengspark-devops - nodeSelector: - dedicated: dengspark-devops diff --git a/helm-overrides/k8s-dengspark-prd-ase1/flagger/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/flagger/custom-values.yaml deleted file mode 100644 index 02a63ec..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/flagger/custom-values.yaml +++ /dev/null @@ -1,55 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: hpa-changes-26 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - dedicated: dengspark-devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dengspark-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: devops - bu: dengspark diff --git a/helm-overrides/k8s-dengspark-prd-ase1/fluentd/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/fluentd/custom-values.yaml deleted file mode 100644 index 3e9307d..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/fluentd/custom-values.yaml +++ /dev/null @@ -1,764 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-dpspsre-fluentd-prd@meesho-dengspark-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: dengspark - team: sre - type: fluentd - service: fluentd-dengspark-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: dengspark - team: sre - type: fluentd - service: fluentd-dengspark-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - - -# externalSecret: -# enabled: false -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: [] -# - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-dengspark-prd-ase1/ingress-nginx-external/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/ingress-nginx-external/custom-values.yaml deleted file mode 100644 index 802f751..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/ingress-nginx-external/custom-values.yaml +++ /dev/null @@ -1,30 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 15 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - ingressClass: nginx-external - ingressClassByName: true - watchIngressWithoutClass: false - ingressClassResource: - controllerValue: k8s.io/ingress-nginx-external - name: nginx-external - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dengspark-external-prd"}}}' - nodeSelector: - dedicated: dengspark-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "dengspark-devops" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-dengspark-prd-ase1/ingress-nginx-internal/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/ingress-nginx-internal/custom-values.yaml deleted file mode 100644 index cb08556..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/ingress-nginx-internal/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 5 - maxReplicas: 15 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dengspark-internal-prd"}}}' - nodeSelector: - dedicated: dengspark-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "dengspark-devops" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-dengspark-prd-ase1/keda/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/keda/custom-values.yaml deleted file mode 100644 index 9d0f70e..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/keda/custom-values.yaml +++ /dev/null @@ -1,45 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: dengspark-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "dengspark-devops" - effect: "NoSchedule" - podLabels: - bu: "dengspark" - team: "dengspark-devops" - metricsAdapter: - bu: "dengspark" - team: "dengspark-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 2000Mi - requests: - cpu: 400m - memory: 300Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 30m - memory: 100Mi - metricServer: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 200m - memory: 200Mi - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' diff --git a/helm-overrides/k8s-dengspark-prd-ase1/kube-events/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/kube-events/custom-values.yaml deleted file mode 100644 index e953887..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-dengspark-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - dedicated: dengspark-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: dengspark-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: dengspark-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: dengspark-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: dengspark-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: dengspark-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-dengspark-prd-ase1/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/kube-state-metrics/custom-values.yaml deleted file mode 100644 index d279481..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-dengspark-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-dengspark-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "kube-state-metrics-dengspark-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "victoriametrics" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 10m - memory: 50Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-dengspark-prd-ase1/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index 1e745fb..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,496 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 200m - # memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-admin-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "node-exporter-dengspark-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-agent/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index 4965462..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,295 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-dengspark-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-dengspark-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dsprk-spsre-vmagent-prd@meesho-dengspark-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-dengspark.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-dengspark-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vmagent-dengspark-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dengspark-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dengspark-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 2 - memory: 4Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "victoriametrics" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml deleted file mode 100644 index 60b140d..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-dengspark-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-dengspark-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-dengspark-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-dengspark-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/dengspark/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-dengspark-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "dengspark" - team: "sre" - service: "vmalert-dengspark-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vminsert" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-alert/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-alert/custom-values.yaml deleted file mode 100644 index 4a40911..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-alert/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-dengspark-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-dengspark-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-dengspark-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-dengspark-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.meeshogcp.in" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config1/dengspark/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-dengspark-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-dengspark-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - dedicated: "vminsert" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-insert/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index be822d1..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,234 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dengspark-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-dengspark-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vminsert-dengspark-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 2 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vminsert" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 2 - memory: 2Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-dengspark-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-select/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index b3ec09d..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,295 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dengspark-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - extraVMSelects: [] - #extraVMSelects: - #- --storageNode=http://vmselect-dengspark-di-prd.meeshogcp.in - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-dengspark-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 96 - clusternative.maxConcurrentRequests: 256 - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vmselect-dengspark-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 20 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "victoriametrics" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 1200m - memory: 4Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-dengspark-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-storage/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index 477fbf0..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,316 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-dengspark-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 30d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 200Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vmstorage-dengspark-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 5 - memory: 25Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.91.3-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-demand-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dengspark-prd-ase1/victoriametrics-agent/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index bb71efb..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,314 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "dengspark-scrape.yaml" - -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -replicaCount: 2 - -fullnameOverride: vm-agent-dengspark-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-dengspark-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.133.0 # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dsprk-spsre-vmagent-prd@meesho-dengspark-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWrite: - - url: http://vm-insert-dengspark-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vm-agent-dengspark-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -podLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vm-agent-dengspark-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dengspark-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dengspark-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 4 - memory: 6Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent-n4d" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-n4d" - effect: "NoSchedule" - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-dengspark-prd-ase1/victoriametrics-insert/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/victoriametrics-insert/custom-values.yaml deleted file mode 100644 index 98233af..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/victoriametrics-insert/custom-values.yaml +++ /dev/null @@ -1,311 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - - - -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-mqkafka-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.107.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-dengspark-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-dengspark-prd-0.vm-storage-dengspark-prd.victoriametrics.svc:8400" - - "vm-storage-dengspark-prd-1.vm-storage-dengspark-prd.victoriametrics.svc:8400" - - "vm-storage-dengspark-prd-2.vm-storage-dengspark-prd.victoriametrics.svc:8400" - - "vm-storage-dengspark-prd-3.vm-storage-dengspark-prd.victoriametrics.svc:8400" - - "vm-storage-dengspark-prd-4.vm-storage-dengspark-prd.victoriametrics.svc:8400" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vm-insert-dengspark-prd" - env: "prd" - priority: "p0" - type: "vminsert" - - podLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vm-insert-dengspark-prd" - env: "prd" - priority: "p0" - type: "vminsert" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 2 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vm-common" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vm-common" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 2 - memory: 2Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-dengspark-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dengspark-prd-ase1/victoriametrics-select/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/victoriametrics-select/custom-values.yaml deleted file mode 100644 index 984e24a..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/victoriametrics-select/custom-values.yaml +++ /dev/null @@ -1,332 +0,0 @@ -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dengspark-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - extraVMSelects: [] - #extraVMSelects: - #- --storageNode=http://vmselect-dengspark-di-prd.meeshogcp.in - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-dengspark-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - # -- Enable clusternative service for LoadBalancer vmselect component - clusternativeService: - enabled: false - externalService: - enabled: true - name: vmselect-dengspark-prd-proxy - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-dengspark-prd-0.vm-storage-dengspark-prd.victoriametrics.svc:8401" - - "vm-storage-dengspark-prd-1.vm-storage-dengspark-prd.victoriametrics.svc:8401" - - "vm-storage-dengspark-prd-2.vm-storage-dengspark-prd.victoriametrics.svc:8401" - - "vm-storage-dengspark-prd-3.vm-storage-dengspark-prd.victoriametrics.svc:8401" - - "vm-storage-dengspark-prd-4.vm-storage-dengspark-prd.victoriametrics.svc:8401" - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vm-select-dengspark-prd" - env: "prd" - priority: "p0" - type: "vmselect" - podLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vm-select-dengspark-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 20 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "vm-common" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vm-common" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 1200m - memory: 5Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vm-select-dengspark-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dengspark-prd-ase1/victoriametrics-storage/custom-values.yaml b/helm-overrides/k8s-dengspark-prd-ase1/victoriametrics-storage/custom-values.yaml deleted file mode 100644 index 9482023..0000000 --- a/helm-overrides/k8s-dengspark-prd-ase1/victoriametrics-storage/custom-values.yaml +++ /dev/null @@ -1,345 +0,0 @@ -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: false - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-dengspark-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 30d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage-n4d" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage-n4d" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 1700Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vm-storage-dengspark-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - podLabels: - bu: "dengspark" - team: "dengspark-sre" - service: "vm-storage-dengspark-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 3 - memory: 25Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.91.3-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-demand-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dscispark-prd-ase1/README.md b/helm-overrides/k8s-dscispark-prd-ase1/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/k8s-dscispark-prd-ase1/conntrack-adjuster/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 7b64b15..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - # Maximum number of conntrack entries - max: 2097152 - # Hash size for conntrack - hashsize: 524288 - # Sleep interval between adjustments (seconds) - sleepInterval: 30 - -# Enable additional matchExpressions -# addExtraMatchExpressions: true - -# Additional matchExpressions to append -# additionalMatchExpressions: -# - key: node-role -# operator: In -# values: -# - "worker" -# - "ingress" -# - key: environment -# operator: In -# values: -# - "production" -# - "staging" \ No newline at end of file diff --git a/helm-overrides/k8s-dscispark-prd-ase1/external-secrets/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/external-secrets/custom-values.yaml deleted file mode 100644 index 887908f..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/external-secrets/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Additional service annotations - annotations: {} - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dscispark-devops - nodeSelector: - dedicated: dscispark-devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dscispark-devops - nodeSelector: - dedicated: dscispark-devops - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dscispark-devops - nodeSelector: - dedicated: dscispark-devops diff --git a/helm-overrides/k8s-dscispark-prd-ase1/flagger/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/flagger/custom-values.yaml deleted file mode 100644 index d87a41c..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/flagger/custom-values.yaml +++ /dev/null @@ -1,55 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: hpa-changes-26 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - dedicated: dscispark-devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dscispark-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: devops - bu: dscispark diff --git a/helm-overrides/k8s-dscispark-prd-ase1/fluentd/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/fluentd/custom-values.yaml deleted file mode 100644 index 4f77e69..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/fluentd/custom-values.yaml +++ /dev/null @@ -1,764 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-dpspsre-fluentd-prd@meesho-dscispark-prd-0125.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: dscispark - team: sre - type: fluentd - service: fluentd-dscispark-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: dscispark - team: sre - type: fluentd - service: fluentd-dscispark-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - - -# externalSecret: -# enabled: false -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: [] -# - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-dscispark-prd-ase1/ingress-nginx-internal/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/ingress-nginx-internal/custom-values.yaml deleted file mode 100644 index ca83ea0..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/ingress-nginx-internal/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 5 - maxReplicas: 15 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dscispark-int-prd"}}}' - nodeSelector: - dedicated: dscispark-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "dscispark-devops" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-dscispark-prd-ase1/keda/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/keda/custom-values.yaml deleted file mode 100644 index fadb604..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/keda/custom-values.yaml +++ /dev/null @@ -1,38 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: dscispark-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "dscispark-devops" - effect: "NoSchedule" - podLabels: - bu: "dscispark" - team: "dscispark-devops" - metricsAdapter: - bu: "dscispark" - team: "dscispark-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 2000Mi - requests: - cpu: 250m - memory: 300Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 30m - memory: 100Mi - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' \ No newline at end of file diff --git a/helm-overrides/k8s-dscispark-prd-ase1/kube-events/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/kube-events/custom-values.yaml deleted file mode 100644 index b9c7ad4..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-dscispark-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - dedicated: dscispark-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: dscispark-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: dscispark-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: dscispark-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: dscispark-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: dscispark-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-dscispark-prd-ase1/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 75aa3cc..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-dscispark-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-dscispark-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dscispark" - team: "dscispark-sre" - service: "kube-state-metrics-dscispark-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "victoriametrics" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 10m - memory: 50Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-dscispark-prd-ase1/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index 2551f47..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,496 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 200m - # memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-admin-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "dscispark" - team: "dscispark-sre" - service: "node-exporter-dscispark-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-agent/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index 7af2f5a..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,295 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-dscispark-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-dscispark-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dsprk-spsre-vmagent-prd@meesho-dscispark-prd-0125.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-dscispark.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-dscispark-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dscispark" - team: "dscispark-sre" - service: "vmagent-dscispark-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dscispark-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dscispark-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 2 - memory: 4Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "victoriametrics" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml deleted file mode 100644 index 43f55d3..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-dscispark-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-dscispark-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-dscispark-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-dscispark-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/dscispark/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-dscispark-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "dscispark" - team: "sre" - service: "vmalert-dscispark-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vminsert" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-alert/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-alert/custom-values.yaml deleted file mode 100644 index 6946199..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-alert/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-dscispark-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-dscispark-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-dscispark-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-dscispark-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.meeshogcp.in" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config1/dscispark/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-dscispark-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-dscispark-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - dedicated: "vminsert" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-insert/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index bb35a28..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,234 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dscispark-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-dscispark-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "dscispark" - team: "dscispark-sre" - service: "vminsert-dscispark-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 2 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vminsert" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 2 - memory: 2Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-dscispark-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-select/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index 102d5c6..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,295 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dscispark-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - extraVMSelects: [] - #extraVMSelects: - #- --storageNode=http://vmselect-dscispark-di-prd.meeshogcp.in - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-dscispark-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 96 - clusternative.maxConcurrentRequests: 256 - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "dscispark" - team: "dscispark-sre" - service: "vmselect-dscispark-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 20 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "victoriametrics" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "victoriametrics" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 1200m - memory: 4Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-dscispark-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-storage/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index 762039d..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,316 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-dscispark-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 30d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage-64g" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage-64g" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 200Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "dscispark" - team: "dscispark-sre" - service: "vmstorage-dscispark-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 5 - memory: 50Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.91.3-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-demand-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dscispark-prd-ase1/victoriametrics-agent/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index 79610b7..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,317 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "dscispark-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. -replicaCount: 2 - -fullnameOverride: vm-agent-dscispark-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-dscispark-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.133.0 # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dsprk-spsre-vmagent-prd@meesho-dscispark-prd-0125.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWrite: - - url: http://vm-insert-dscispark-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "dscispark" - team: "dscispark-sre" - service: "vm-agent-dscispark-prd" - env: "prd" - priority: "p0" - type: "vmagent" -podLabels: - bu: "dscispark" - team: "dscispark-sre" - service: "vm-agent-dscispark-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-dscispark-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: false - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-dscispark-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 4 - memory: 6Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent-n4d" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-n4d" - effect: "NoSchedule" - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-dscispark-prd-ase1/victoriametrics-insert/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/victoriametrics-insert/custom-values.yaml deleted file mode 100644 index 9f49329..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/victoriametrics-insert/custom-values.yaml +++ /dev/null @@ -1,267 +0,0 @@ -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dscispark-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.107.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-dscispark-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-dscispark-prd-0.vm-storage-dscispark-prd.victoriametrics.svc:8400" - - "vm-storage-dscispark-prd-1.vm-storage-dscispark-prd.victoriametrics.svc:8400" - - "vm-storage-dscispark-prd-2.vm-storage-dscispark-prd.victoriametrics.svc:8400" - - "vm-storage-dscispark-prd-3.vm-storage-dscispark-prd.victoriametrics.svc:8400" - - "vm-storage-dscispark-prd-4.vm-storage-dscispark-prd.victoriametrics.svc:8400" - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "dscispark" - team: "dscispark-sre" - service: "vm-insert-dscispark-prd" - env: "prd" - priority: "p0" - type: "vminsert" - podLabels: - bu: "dscispark" - team: "dscispark-sre" - service: "vm-insert-dscispark-prd" - env: "prd" - priority: "p0" - type: "vminsert" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 2 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vm-common" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vm-common" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 2 - memory: 2Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vm-insert-dscispark-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dscispark-prd-ase1/victoriametrics-select/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/victoriametrics-select/custom-values.yaml deleted file mode 100644 index 15da560..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/victoriametrics-select/custom-values.yaml +++ /dev/null @@ -1,331 +0,0 @@ -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-dscispark-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - clusternativeService: - enabled: false - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - extraVMSelects: [] - #extraVMSelects: - #- --storageNode=http://vmselect-dscispark-di-prd.meeshogcp.in - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-dscispark-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - externalService: - enabled: true - name: vmselect-dscispark-prd-proxy - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-dscispark-prd-0.vm-storage-dscispark-prd.victoriametrics.svc:8401" - - "vm-storage-dscispark-prd-1.vm-storage-dscispark-prd.victoriametrics.svc:8401" - - "vm-storage-dscispark-prd-2.vm-storage-dscispark-prd.victoriametrics.svc:8401" - - "vm-storage-dscispark-prd-3.vm-storage-dscispark-prd.victoriametrics.svc:8401" - - "vm-storage-dscispark-prd-4.vm-storage-dscispark-prd.victoriametrics.svc:8401" - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "dscispark" - team: "dscispark-sre" - service: "vm-select-dscispark-prd" - env: "prd" - priority: "p0" - type: "vmselect" - podLabels: - bu: "dscispark" - team: "dscispark-sre" - service: "vm-select-dscispark-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 20 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "vm-common" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vm-common" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 1200m - memory: 3Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vm-select-dscispark-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dscispark-prd-ase1/victoriametrics-storage/custom-values.yaml b/helm-overrides/k8s-dscispark-prd-ase1/victoriametrics-storage/custom-values.yaml deleted file mode 100644 index ce83e0d..0000000 --- a/helm-overrides/k8s-dscispark-prd-ase1/victoriametrics-storage/custom-values.yaml +++ /dev/null @@ -1,345 +0,0 @@ -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: false - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-dscispark-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 30d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage-n4d" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage-n4d" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 3800Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "dscispark" - team: "dscispark-sre" - service: "vm-storage-dscispark-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - podLabels: - bu: "dscispark" - team: "dscispark-sre" - service: "vm-storage-dscispark-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 5 - memory: 50Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.91.3-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-demand-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/README.md b/helm-overrides/k8s-dsgpu-prd-ase1/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/aurva-dataplane/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/aurva-dataplane/custom-values.yaml deleted file mode 100644 index a396b09..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/aurva-dataplane/custom-values.yaml +++ /dev/null @@ -1,660 +0,0 @@ -## Aurva Data Plane -## Ref: https://github.com/aurva-io/aurva-charts.git - -postgresql: - enabled: true - fullnameOverride: "aurva-dataplane-database" - volumePermissions: - ## @param volumePermissions.enabled Enable init container that changes the owner and group of the persistent volume - ## - enabled: true - global: - storageClass: pd-standard-retain - postgresql: - auth: - postgresPassword: "aurva" - database: "controller" - # Add toleration to make sure where this postgres db pod should reside (Applicable for production workloads): For more detail ref: https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/ - primary: - extendedConfiguration: | - max_connections = 300 - - # #PLACEHOLDER## - # tolerations: - # - effect: NoSchedule - # key: cloud.google.com/compute-class - # operator: Equal - # value: datascience-devops - - # # -- Select nodes to deploy which matches the following labels - # nodeSelector: ##PLACEHOLDER## - # dedicated: datascience-devops - -# -- Provide a name in place of `aurva` -# namespaceOverride: aurva-dataplane - -########################################################## -# Global Configs -########################################################## -global: - aurva_controller: - enabled: true - aurva_fastdet: - enabled: false - aurva_pii_analyzer: - enabled: true - aurva_ocr: - enabled: true - aurva_collector: - enabled: true - - deploymentAnnotations: {} - - priorityClassName: "" - -########################################################## -# Aurva Controller -########################################################## -aurva_controller: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "dsgpu" - team: "dsgpu-devops" - service: "aurva-dsgpu-prd" - env: "prd" - priority: "p0" - type: "aurva_controller" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 10 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - image: - # Image of the app container - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-controller/aurva-controller - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - - type: secret - name: aurva-controller-secrets - # - type: configmap - # name: proxy-datasource-config - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 2 - requests: - memory: 1Gi - cpu: 1 - - aurvaFastdet: - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-fastdet/aurva-fastdet - tag: "v2.30.13" - pullPolicy: IfNotPresent - envFrom: [] - env: [] - resources: - limits: - cpu: 0.5 - memory: 512Mi - requests: - cpu: 0.5 - memory: 512Mi - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - - # -- Taint tolerations for nodes - ##PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: e2-standard-4-datascience-devops-compute-class - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-controller-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - config: - #variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - COMMAND_URL: "command.aurva-prd.meeshogcp.in:80" - DEPLOYMENT_TYPE: "kubernetes" - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - FLUSHER_WORKER_POOL_SIZE: "1000" - FLUSHER_BATCH_SIZE: "30000" - UNIQUENESS_IDENTIFIER: "k8s-dsgpu-prd-ase1" #Recommendation: should be equal to cluster name - PROVIDER_ACCOUNT_ID: "meesho-datascience-prd-0622" # GCP PROJECT ID (not Number) - REGION: "asia-southeast1" #eg: asia-south1 - ENVIRONMENT: "prod" - OCR_ENABLED: "true" - MONITORING_ENABLED: "false" - HYBRID_ONLY_MODE: "false" - FORCE_TLS: "false" - FASTDET_FLAG : "true" - AADHAAR_ENHANCER: "0" - WORKSPACE_EVENT_TRACKING_ENABLED: "false" - ACCESS_IQ_ENABLED: "true" - GRPC_ENFORCE_ALPN_ENABLED: "false" - ENABLE_FASTDET: "true" - GOGC: "70" - #pii data - PII_BUCKET_NAME: "gcs-infra-devop-aurva-dsgpu-prd" - PII_LOG_BUCKET_REGION: "asia-southeast1" - PII_LOG_CRON: "*/5 * * * *" - ENABLE_PII_LOG: "true" - PII_EVIDENCE_UPLOAD_MAX_BLOCKING_TASKS: "50000" - PII_EVIDENCE_UPLOAD_MAX_CONCURRENT_TASKS: "500" - PII_EVIDENCE_MAX_CACHE_WEIGHT: "100" - QUOTA_CLEANUP_CRON: "0 0 * * *" - ENABLE_PII_QUOTA: "true" - MAX_PII_EVIDENCES_PER_KEY_PER_WINDOW: "3" - PII_QUOTA_SYNC_CRON: "*/2 * * * *" - QUOTA_WINDOW_HOURS: "24" - #constants - SKIP_NAMESPACES: "contour-internal-1-dsgpu-prd,contour-internal-0-dsgpu-prd,contour-external-dsgpu-prd" - CLOUD_PROVIDER: "gcp" - LOG_ENV: "production" - RDS_SCANNER_AVAILABILITY : "false" - REDSHIFT_SCANNER_AVAILABILITY : "false" - S3_SCANNER_AVAILABILITY : "false" - DYNAMO_SCANNER_AVAILABILITY: "false" - DOCDB_SCANNER_AVAILABILITY: "false" - OPENSEARCH_SCANNER_AVAILABILITY: "false" - CLOUDSQL_SCANNER_AVAILABILITY: "true" - BIGQUERY_SCANNER_AVAILABILITY: "true" - AWS_SNAPSHOT_SCANNER_AVAILABILITY: "false" - CLOUDSTORAGE_SCANNER_AVAILABILITY: "true" - KEYSPACES_SCANNER_AVAILABILITY: "false" - ALLOYDB_SCANNER_AVAILABILITY: "true" - BIGTABLE_SCANNER_AVAILABILITY: "true" - GCP_BACKUP_AVAILABILITY: "true" - EGRESS_MODE_ONLY: "false" - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - SCAN_UUID_ENABLED: "true" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-controller-sa - # -- Annotations applied to created service account - annotations: - iam.gke.io/gcp-service-account: sa-datascience-prd-aurva-contr@meesho-datascience-prd-0622.iam.gserviceaccount.com -# eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 15 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 - - -########################################################## -# Aurva OCR -########################################################## -aurva_ocr: - # -- Additional labels for aurva-controller - additionalLabels: - bu: "dsgpu" - team: "dsgpu-devops" - service: "aurva-dsgpu-prd" - env: "prd" - priority: "p0" - type: "aurva_ocr" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 1 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-ocr/aurva-ocr - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-ocr: - type: secret - name: aurva-ocr-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 1 - requests: - memory: 2Gi - cpu: 1 - - # -- Select nodes to deploy which matches the following labels - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: e2-standard-4-datascience-devops-compute-class - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-ocr-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - OCR_TIME_LIMIT: "1" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-ocr-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # iam.gke.io/gcp-service-account: service-account@gcp.iam.gserviceaccount.com - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva Collector -########################################################## -aurva_collector: - - # -- Additional labels for aurva-analyzer - additionalLabels: - bu: "dsgpu" - team: "dsgpu-devops" - service: "aurva-dsgpu-prd" - env: "prd" - priority: "p0" - type: "aurva_collector" - - - # -- Annotations on aurva-analyzer - annotations: {} -# "key": "value" - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-collector/aurva-collector - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-controller: - type: secret - name: aurva-collector-secrets - - resources: - limits: - cpu: 800m - memory: 800Mi - requests: - cpu: 200m - memory: 512Mi - - podSecurityContext: {} - - securityContext: - privileged: false - capabilities: - add: - # For kernel v5.8 and above we don't need CAP_SYS_ADMIN or CAP_SYS_RESOURCE - # we just need CAP_BPF and CAP_PERFMON. This has been tested on our EKS node - # which is on kernel v5.10.x - # When SSL Tracing is required we need CAP_SYS_ADMIN and CAP_SYS_PTRACE - # on top of the previous capabilities - # So finally these are the 4 possible combinations for capabilies - # 1. Newer Kernels without SSL - - BPF - - PERFMON - # 2. Newer Kernels with SSL - # - SYS_ADMIN - # - SYS_PTRACE - # 3. Older Kernels without SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # 4. Older Kernels with SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # - SYS_PTRACE - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - volumes: - - name: debugfs - mountPath: /sys/kernel/debug - hostPath: /sys/kernel/debug - - name: vmlinux - mountPath: /sys/kernel/btf/vmlinux - hostPath: /sys/kernel/btf/vmlinux - - name: procfs - mountPath: /host/proc - hostPath: /proc - - # -- Taint tolerations for nodes - tolerations: - # - effect: NoSchedule - # key: cloud.google.com/compute-class - # operator: Equal - # value: megatetra - - operator: Exists - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: cloud.google.com/compute-class - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external - - alloy - - preprod-spot-16 - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-collector-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - # variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - TRACE_INTERNAL_SVC: "true" - TRACE_INTERNAL_SVC_HTTP: "true" - INTERNAL_SVC_SAMPLE_INTERVAL: "2m" - TRACE_SSL: "false" - TRACE_PSQL: "false" - TRACE_SQLSERVER: "false" - TRACE_MYSQL: "false" - TRACE_EGRESS: "true" - TRACE_GO_TLS: "false" - TRACE_ML_SERVICES: "false" - # constants - LOG_ENV: production - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - MONITORING_ENABLED: "false" - ENABLE_INGRESS_INFORMER: "false" - ENABLE_SERVICE_INFORMER: "false" - ENABLE_ISTIO_INFORMER: "false" - EXCLUDED_PII_REGEX_TYPES: "ip_address,us_bank_number,us_driver_license,us_itin,us_passport,us_routing,us_mbi,ssn" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-collector-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva PII Analyzer -########################################################## -aurva_pii_analyzer: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "dsgpu" - team: "dsgpu-devops" - service: "aurva-dsgpu-prd" - env: "prd" - priority: "p0" - type: "aurva_pii_analyzer" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 3 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - ##PLACEHOLDER## - nodeSelector: - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-piianalyzer/aurva-piianalyzer - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-pii-analyzer: - type: secret - name: aurva-pii-analyzer-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 4Gi - cpu: 4 - requests: - memory: 2Gi - cpu: 2 - - nodeSelector: ##PLACEHOLDER## - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: e2-standard-4-datascience-devops-compute-class - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-pii-analyzer-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - SCHEDULER_TIME: "1" - SUPPORTED_REGION: "US" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-pii-analyzer-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 3 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/cert-manager/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/cert-manager/custom-values.yaml deleted file mode 100644 index 6ee0484..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/cert-manager/custom-values.yaml +++ /dev/null @@ -1,129 +0,0 @@ -global: - logLevel: 2 - rbac: - create: true - priorityClassName: "high-priority" -installCRDs: false - -crds: - enabled: true - keep: true - -# Cert-manager Controller -replicaCount: 1 -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-controller - tag: v1.20.1 - pullPolicy: IfNotPresent - -nodeSelector: - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "e2-standard-4-datascience-devops-compute-class" - effect: "NoSchedule" - -resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 500m - memory: 512Mi - -# Webhook Configuration -webhook: - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-webhook - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "e2-standard-4-datascience-devops-compute-class" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# CA Injector Configuration -cainjector: - enabled: true - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-cainjector - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "e2-standard-4-datascience-devops-compute-class" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# ACME Solver Configuration -acmesolver: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-acmesolver - tag: v1.20.1 - pullPolicy: IfNotPresent - -# Startup API Check -startupapicheck: - enabled: true - timeout: 1m - backoffLimit: 4 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-startupapicheck - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "e2-standard-4-datascience-devops-compute-class" - effect: "NoSchedule" - - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - -# Prometheus Monitoring -prometheus: - enabled: true - servicemonitor: - enabled: false - interval: 60s - scrapeTimeout: 30s - labels: - prometheus: cert-manager diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/computeclass/g2-standard-16-l4-compute-class.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/computeclass/g2-standard-16-l4-compute-class.yaml deleted file mode 100644 index 90e49bf..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/computeclass/g2-standard-16-l4-compute-class.yaml +++ /dev/null @@ -1,37 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-16-l4-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-a - machineType: g2-standard-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-b - machineType: g2-standard-16 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/computeclass/g2-standard-4-l4-compute-class.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/computeclass/g2-standard-4-l4-compute-class.yaml deleted file mode 100644 index 68f50eb..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/computeclass/g2-standard-4-l4-compute-class.yaml +++ /dev/null @@ -1,37 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-4-l4-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-a - machineType: g2-standard-4 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-b - machineType: g2-standard-4 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/computeclass/g2-standard-8-l4-300gb-compute-class.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/computeclass/g2-standard-8-l4-300gb-compute-class.yaml deleted file mode 100644 index 6ce5c63..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/computeclass/g2-standard-8-l4-300gb-compute-class.yaml +++ /dev/null @@ -1,39 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-8-l4-300gb-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-a - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskSize: 300 - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-b - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskSize: 300 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/computeclass/g2-standard-8-l4-compute-class.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/computeclass/g2-standard-8-l4-compute-class.yaml deleted file mode 100644 index afdc331..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/computeclass/g2-standard-8-l4-compute-class.yaml +++ /dev/null @@ -1,37 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: g2-standard-8-l4-compute-class -spec: - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - nodePoolConfig: - serviceAccount: gke-gpu-node-sa@meesho-datascience-prd-0622.iam.gserviceaccount.com - priorities: - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-a - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - - gpu: - count: 1 - driverVersion: default - type: nvidia-l4 - location: - zones: - - asia-southeast1-b - machineType: g2-standard-8 - maxPodsPerNode: 18 - spot: false - storage: - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/conntrack-adjuster/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 993f9c4..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - # Maximum number of conntrack entries - max: 2097152 - # Hash size for conntrack - hashsize: 524288 - # Sleep interval between adjustments (seconds) - sleepInterval: 30 - -# Enable additional matchExpressions -addExtraMatchExpressions: true - -# Additional matchExpressions to append -additionalMatchExpressions: - - key: dedicated - operator: In - values: - - "megaquad" -# - "ingress" -# - key: environment -# operator: In -# values: -# - "production" -# - "staging" diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/contour-ca-issuer/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/contour-ca-issuer/custom-values.yaml deleted file mode 100644 index 73370d1..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/contour-ca-issuer/custom-values.yaml +++ /dev/null @@ -1,17 +0,0 @@ -# ClusterIssuer override for k8s-dsgpu-prd-ase1 (prd dsgpu cluster, ase1a zone). -# ClusterIssuer is a cluster-scoped resource — this file pins the issuer -# identity for this cluster so the name is auditable per-cluster. -# -# Must match the `issuerRef.name` in consuming Certificate CRs (see -# devops-helm-charts/2.0.0/templates/proxyless-grpc-cert.yaml). - -issuerName: contour-dsgpu-prd-ca-issuer -rootCASecretName: contour-dsgpu-ca - -externalSecret: - enabled: true - vaultPath: meesho/devops/contour/root-ca - refreshInterval: "0" - secretStoreRef: - name: vault-backend -namespace: cert-manager-dsgpu-prd \ No newline at end of file diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/contour-cert-checker/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/contour-cert-checker/custom-values.yaml deleted file mode 100644 index c61998a..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=k8s-dsgpu-prd-ase1"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: e2-standard-4-datascience-devops-compute-class - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/contour-internal-0/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/contour-internal-0/custom-values.yaml deleted file mode 100644 index e62a8d7..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,102 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 60s - max-connection-duration: 600s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: dsgpu - team: dsgpu-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 500m - memory: 3Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: c3-highcpu-22-contour-internal-0-compute-class - nodeSelector: - cloud.google.com/compute-class: c3-highcpu-22-contour-internal-0-compute-class - service: - tcpLB: true - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: dsgpu - team: dsgpu-devops - env: prd - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 5% - maxUnavailable: 0 - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: c3-highcpu-22-contour-internal-0-compute-class - nodeSelector: - cloud.google.com/compute-class: c3-highcpu-22-contour-internal-0-compute-class - logLevel: error - extraArgs: - - '--concurrency 8' - autoscaling: - enabled: true - minReplicas: 9 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 12Gi - limits: - cpu: 8 - memory: 12Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-dsgpu-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/contour-internal-1/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/contour-internal-1/custom-values.yaml deleted file mode 100644 index 083e4e2..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,112 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 400s - max-connection-duration: 540s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: dsgpu - team: dsgpu-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 8 - memory: 10Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: c4d-highcpu-16-contour-internal-1-compute-class - nodeSelector: - cloud.google.com/compute-class: c4d-highcpu-16-contour-internal-1-compute-class - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: dsgpu - team: dsgpu-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: c4d-highcpu-16-contour-internal-1-compute-class - nodeSelector: - cloud.google.com/compute-class: c4d-highcpu-16-contour-internal-1-compute-class - logLevel: error - terminationGracePeriodSeconds: 500 - autoscaling: - behaviour: - scaledown: - policies: - - periodseconds: 500 - type: Pods - value: 1 - selectpolicy: Min - stabilizationWindowSeconds: 600 - scaleup: - policies: - - periodseconds: 15 - type: Pods - value: 2 - selectpolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "35" - targetMemory: "40" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 14 - memory: 8Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int1-dsgpu-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/contour-internal-intra-0/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index a6ccd8d..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,95 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 60s - max-connection-duration: 600s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: dsgpu - team: dsgpu-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 500m - memory: 3Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: c4d-highcpu-16-contour-internal-0-compute-class - nodeSelector: - cloud.google.com/compute-class: c4d-highcpu-16-contour-internal-0-compute-class - service: - tcpLB: true - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: dsgpu - team: dsgpu-devops - env: prd - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 5% - maxUnavailable: 0 - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: c4d-highcpu-16-contour-internal-0-compute-class - nodeSelector: - cloud.google.com/compute-class: c4d-highcpu-16-contour-internal-0-compute-class - logLevel: error - autoscaling: - enabled: true - minReplicas: 9 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 14 - memory: 8Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/contour-internal-intra-1/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index 8d8cf02..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,110 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 400s - max-connection-duration: 540s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: dsgpu - team: dsgpu-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 8 - memory: 10Gi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: c3-highcpu-22-contour-internal-1-compute-class - nodeSelector: - cloud.google.com/compute-class: c3-highcpu-22-contour-internal-1-compute-class - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: dsgpu - team: dsgpu-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: c3-highcpu-22-contour-internal-1-compute-class - nodeSelector: - cloud.google.com/compute-class: c3-highcpu-22-contour-internal-1-compute-class - logLevel: error - terminationGracePeriodSeconds: 500 - autoscaling: - behaviour: - scaledown: - policies: - - periodseconds: 500 - type: Pods - value: 1 - selectpolicy: Min - stabilizationWindowSeconds: 600 - scaleup: - policies: - - periodseconds: 15 - type: Pods - value: 2 - selectpolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 10 - maxReplicas: 250 - targetCPU: "35" - targetMemory: "40" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 18 - memory: 12Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/coredns/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/coredns/custom-values.yaml deleted file mode 100644 index bda7c29..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/coredns/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -replicaCount: 30 -communicationType: "intra" - -labels: - bu: dsgpu - team: datascience-devops - env: prd - -clusterIP: 10.180.64.2 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "e2-standard-4-datascience-devops-compute-class" - effect: "NoSchedule" - -nodeSelector: - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - kubernetes.io/os: linux diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/external-dns-services/contour-internal-0-dsgpu-prd-envoy-headless-external-dns-svc.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/external-dns-services/contour-internal-0-dsgpu-prd-envoy-headless-external-dns-svc.yaml deleted file mode 100644 index bb75c04..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/external-dns-services/contour-internal-0-dsgpu-prd-envoy-headless-external-dns-svc.yaml +++ /dev/null @@ -1,30 +0,0 @@ -apiVersion: v1 -kind: Service -metadata: - annotations: - external-dns.alpha.kubernetes.io/hostname: contour-internal-0-dsgpu.dsgpu.meesho.int - labels: - app.kubernetes.io/component: envoy - app.kubernetes.io/instance: contour-internal-0-dsgpu-prd - app.kubernetes.io/managed-by: Helm - app.kubernetes.io/name: contour - app.kubernetes.io/version: 1.27.0 - argocd.argoproj.io/instance: contour-internal-0-dsgpu-prd - helm.sh/chart: contour-15.0.0 - name: contour-internal-0-dsgpu-prd-envoy-headless-external-dns-svc - namespace: contour-internal-0-dsgpu-prd -spec: - clusterIP: None - clusterIPs: - - None - ipFamilies: - - IPv4 - ipFamilyPolicy: SingleStack - ports: - - name: http - port: 80 - targetPort: http - selector: - app.kubernetes.io/component: envoy - app.kubernetes.io/instance: contour-internal-0-dsgpu-prd - app.kubernetes.io/name: contour diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/external-dns/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/external-dns/custom-values.yaml deleted file mode 100644 index 125029e..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/external-dns/custom-values.yaml +++ /dev/null @@ -1,60 +0,0 @@ -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/external-dns - tag: "v0.13.4" - -fullnameOverride: prd-external-dns - -serviceAccount: - create: true - name: prd-external-dns - annotations: - iam.gke.io/gcp-service-account: sa-admin-external-dns-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - -provider: - name: google - -sources: - - service - -policy: upsert-only - -registry: txt -txtOwnerId: gke-externaldns-dsgpu-prd - -domainFilters: - - dsgpu.meesho.int - -interval: 10s - -logLevel: info -logFormat: text - -revisionHistoryLimit: 10 - -deploymentStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 0 - -priorityClassName: "high-priority" - -terminationGracePeriodSeconds: 30 -dnsPolicy: ClusterFirst - -podAnnotations: - cluster-autoscaler.kubernetes.io/safe-to-evict: "false" - -podLabels: - app: external-dns - -extraArgs: - google-batch-change-size: "1000" - google-project: meesho-admin-prd-0622 - -rbac: - create: true - additionalPermissions: - - apiGroups: [""] - resources: ["endpoints"] - verbs: ["get", "watch", "list"] diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/external-secrets/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/external-secrets/custom-values.yaml deleted file mode 100644 index d31e693..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/external-secrets/custom-values.yaml +++ /dev/null @@ -1,61 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Annotations to add to Pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: e2-standard-4-datascience-devops-compute-class - nodeSelector: - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - certController: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: e2-standard-4-datascience-devops-compute-class - nodeSelector: - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - webhook: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: e2-standard-4-datascience-devops-compute-class - nodeSelector: - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" \ No newline at end of file diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/flagger/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/flagger/custom-values.yaml deleted file mode 100644 index b991f0c..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/flagger/custom-values.yaml +++ /dev/null @@ -1,62 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - -tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: e2-standard-4-datascience-devops-compute-class - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: dsgpu-devops - bu: infra - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-dsgpu-rollout-service.prd-dsgpu-rollout-service.svc.cluster.local" - -rolloutService: - enabled: true - apps: diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/fluentd/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/fluentd/custom-values.yaml deleted file mode 100644 index 7fc5c04..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/fluentd/custom-values.yaml +++ /dev/null @@ -1,731 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-dssre-fluentd-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: dsgpu - team: sre - type: fluentd - service: fluentd-dsgpu-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: dsgpu - team: sre - type: fluentd - service: fluentd-dsgpu-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: - - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - 04_outputs.conf: |- - diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/ingress-nginx-internal/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/ingress-nginx-internal/custom-values.yaml deleted file mode 100644 index 2345c03..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/ingress-nginx-internal/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 6 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-dsgpu-int-prd"}}}' - nodeSelector: - cloud.google.com/compute-class: c3-highcpu-4-nginx-internal-compute-class - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "c3-highcpu-4-nginx-internal-compute-class" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/keda/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/keda/custom-values.yaml deleted file mode 100644 index 3fedf7d..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/keda/custom-values.yaml +++ /dev/null @@ -1,46 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "e2-standard-4-datascience-devops-compute-class" - effect: "NoSchedule" - podLabels: - bu: "dsgpu" - team: "dsgpu-devops" - metricsAdapter: - bu: "dsgpu" - team: "dsgpu-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 300m - memory: 450Mi - # -- Manage [resource request & limits] of KEDA metrics apiserver pod - metricServer: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 100m - memory: 200Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 100m - memory: 180Mi - requests: - cpu: 60m - memory: 120Mi - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' \ No newline at end of file diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/kube-dns/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/kube-dns/custom-values.yaml deleted file mode 100644 index ca6a48e..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.180.64.2"],"prd.mrouter.int.svc.cluster.local":["10.180.64.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.180.64.2"]} diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/kube-events/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/kube-events/custom-values.yaml deleted file mode 100644 index 8fc0851..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/kube-events/custom-values.yaml +++ /dev/null @@ -1,148 +0,0 @@ -fullnameOverride: kube-events-dsgpu-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: e2-standard-4-datascience-devops-compute-class - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: e2-standard-4-datascience-devops-compute-class - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: e2-standard-4-datascience-devops-compute-class - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/kube-state-metrics/custom-values.yaml deleted file mode 100644 index ec44293..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-dsgpu-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-dsgpu-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "kube-state-metrics-dsgpu-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - cloud.google.com/compute-class: "c3-highcpu-44-vmselect-compute-class" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "c3-highcpu-44-vmselect-compute-class" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 17m - memory: 1.5Gi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap:z -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/kubectl-mcp-server/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/kubectl-mcp-server/custom-values.yaml deleted file mode 100644 index c781304..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/kubectl-mcp-server/custom-values.yaml +++ /dev/null @@ -1,121 +0,0 @@ -fullnameOverride: "kubectl-mcp-server" - -replicas: 1 - -image: - # In-house hardened build (helm-templates/kubectl-mcp-server/docker/) — - # NOT the upstream Docker Hub image. 266→18 HIGH/CRIT vulns, non-root, - # multi-stage slim base, kubectl v1.33.12 / helm v3.21.0. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/prd/devop/kubectl-mcp-server - tag: "v2" - pullPolicy: IfNotPresent - -labels: - bu: infra - team: devops - service: kubectl-mcp-server - env: prd - -serviceAccount: - create: true - annotations: {} - -rbac: - create: true - -podAnnotations: {} - -podSecurityContext: {} - -securityContext: - runAsNonRoot: true - runAsUser: 1000 - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -priorityClassName: "" - -# dsgpu is a GPU cluster with no `dedicated` nodepools — it uses GKE custom -# compute classes (NAP). The devops workload class is -# `e2-standard-4-datascience-devops-compute-class`; nodes carry both the -# label and a matching NoSchedule taint (verified on k8s-dsgpu-prd-ase1). -nodeSelector: - cloud.google.com/compute-class: "e2-standard-4-datascience-devops-compute-class" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "e2-standard-4-datascience-devops-compute-class" - effect: NoSchedule - -affinity: {} - -resources: - requests: - cpu: 250m - memory: 512Mi - limits: - cpu: 1000m - memory: 1Gi - -mcp: - mode: single - transport: http - host: "0.0.0.0" - port: 8000 - -auth: - allowAnonymous: false - -externalSecrets: - enabled: true - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/prd/cntr/devop/kubectl-mcp-server-dsgpu" - -# tcpSocket on purpose — streamable-http transport exposes only /mcp, -# there is no /health route (see chart values.yaml note). -livenessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - -readinessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - -service: - port: 8000 - type: ClusterIP - -# Contour standard (2.0.0-style): renders HTTPProxy parent/child + -intra -# instead of a networking.k8s.io Ingress. -createContourGateway: true - -ingress: - enabled: true - ingressClassName: contour-internal-1 - servicePortNumber: 8000 - hosts: - - host: kubectl-mcp-server-dsgpu.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/loadtester/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/loadtester/custom-values.yaml deleted file mode 100644 index 5d18817..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/loadtester/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -replicaCount: 1 - -Namespace: loadtester - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/dev/devops/flagger-loadtester - tag: 0.30.0 - pullPolicy: IfNotPresent - pullSecret: - -podLabels: - bu: dsgpu - team: dsgpu-devops - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - -podPriorityClassName: "" - -logLevel: info -cmd: - timeout: 1h - namespaceRegexp: "" - -nameOverride: "loadtester" -fullnameOverride: "" - -env: [] - -service: - type: ClusterIP - port: 80 - -resources: - requests: - cpu: 10m - memory: 64Mi - -volumes: [] -volumeMounts: [] - -nodeSelector: - cloud.google.com/compute-class: e2-standard-4-datascience-devops-compute-class - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "e2-standard-4-datascience-devops-compute-class" - effect: "NoSchedule" - -affinity: {} - -ingress: - slowStart: - enabled: false - enabled: true - ingressClassName: contour-internal-1 - servicePort: http - hosts: - - host: dsgpu-flagger-loadtester.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / -createContourGateway: true -contourResponseTimeout: false -rbac: - # rbac.create: `true` if rbac resources should be created - create: true - # rbac.scope: `cluster` to create cluster-scope rbac resources (ClusterRole/ClusterRoleBinding) - # otherwise, namespace-scope rbac resources will be created (Role/RoleBinding) - scope: - # rbac.rules: array of rules to apply to the role. example: - # rules: - # - apiGroups: [""] - # resources: ["pods"] - # verbs: ["list", "get"] - rules: [] - -# name of an existing service account to use - if not creating rbac resources -serviceAccountName: "" - -# App Mesh virtual node settings (to be used for AppMesh v1beta1) -meshName: "" -#backends: -# - app1.namespace -# - app2.namespace - -# App Mesh virtual node settings (to be used for AppMesh v1beta2) -appmesh: - enabled: false - backends: - - podinfo - - podinfo-canary - -#Istio virtual service and gatway settings. TLS secrets should be in namespace before enbaled it. ( secret format loadtester.fullname ) -istio: - enabled: false - host: flagger-loadtester.flagger - gateway: - enabled: false - tls: - enabled: false - httpsRedirect: false - -# when enabled, it will add a security context for the loadtester pod -securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 100 - runAsGroup: 101 - -podDisruptionBudget: - enabled: false - minAvailable: 1 diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/opentelemetry-coralogix/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/opentelemetry-coralogix/custom-values.yaml deleted file mode 100644 index d96164d..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/opentelemetry-coralogix/custom-values.yaml +++ /dev/null @@ -1,281 +0,0 @@ -global: - domain: "coralogixsg.com" - defaultApplicationName: "default" - defaultSubsystemName: "nodes" - - # Old endpoint based configuration, - # please use domain instead. - traces: - endpoint: "" - metrics: - endpoint: "" - logs: - endpoint: "" - -# set distribution to openshift for openshift clusters -distribution: "" - -# Opentelemetry collector configuration -dedicatedValue: false -schedulerName: default-scheduler - -labels: - bu: "dsgpu" - team: "sre" - service: "opentelemetry-dsgpu-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: false - key: prd/admin/coralogix-keys - secretStoreRef: - name: vault-backend - -image: - # If you want to use the core image `otel/opentelemetry-collector`, you also need to change `command.name` value to `otelcol`. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - # repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - tag: "0.87.0" - # When digest is set to a non-empty value, images will be pulled by digest (regardless of tag value). - digest: "" -mode: daemonset -rollout: - rollingUpdate: - # When 'mode: daemonset', maxSurge cannot be used when hostPort is set for any of the ports - # maxSurge: 25% - maxUnavailable: 1 - strategy: RollingUpdate -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" -hostNetwork: true -dnsPolicy: "ClusterFirstWithHostNet" -fullnameOverride: opentelemetry-dsgpu-prd - -presets: - logsCollection: - enabled: false - storeCheckpoints: false - maxRecombineLogSize: 1048576 - extraFilelogOperators: [] -# - type: recombine -# combine_field: body -# source_identifier: attributes["log.file.path"] -# is_first_entry: body matches "^(YOUR-LOGS-REGEX)" - kubernetesAttributes: - enabled: false - hostMetrics: - enabled: false - kubeletMetrics: - enabled: false - -extraEnvs: -- name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" -- name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName -config: - extensions: {} - # zpages: - # endpoint: localhost:55679 - # pprof: - # endpoint: localhost:1777 - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 4s - tls: - insecure: true - sending_queue: - enabled: true - queue_size: 6000 - num_consumers: 200 - resolver: - dns: - hostname: opentelemetry-admin-prd.opentelemetry.svc.clusterset.local - # coralogix: - # timeout: "30s" - # private_key: "${CORALOGIX_PRIVATE_KEY}" - # domain: "{{.Values.global.domain}}" - # traces: - # endpoint: "{{ .Values.global.traces.endpoint }}" - # metrics: - # endpoint: "{{ .Values.global.metrics.endpoint }}" - # logs: - # endpoint: "{{ .Values.global.logs.endpoint }}" - # application_name_attributes: - # - "k8s.namespace.name" - # - "service.namespace" - # subsystem_name_attributes: - # - "k8s.deployment.name" - # - "k8s.statefulset.name" - # - "k8s.daemonset.name" - # - "k8s.cronjob.name" - # - "k8s.job.name" - # - "k8s.container.name" - # - "k8s.node.name" - # - "service.name" - # application_name: "{{.Values.global.defaultApplicationName }}" - # subsystem_name: "{{.Values.global.defaultSubsystemName }}" - processors: - k8sattributes: - filter: - node_from_env_var: KUBE_NODE_NAME - extract: - metadata: - - "k8s.namespace.name" - - "k8s.deployment.name" - - "k8s.statefulset.name" - - "k8s.daemonset.name" - - "k8s.cronjob.name" - - "k8s.job.name" - - "k8s.pod.name" - - "k8s.node.name" - memory_limiter: null # Will get the k8s resource limits - resourcedetection/env: - detectors: ["system","env"] - timeout: 2s - override: false - # spanmetrics: - # metrics_exporter: coralogix - # dimensions: - # - name: "k8s.deployment.name" - # - name: "k8s.statefulset.name" - # - name: "k8s.daemonset.name" - # - name: "k8s.cronjob.name" - # - name: "k8s.job.name" - # - name: "k8s.container.name" - # - name: "k8s.node.name" - # - name: "k8s.namespace.name" - receivers: - otlp: - protocols: - grpc: - endpoint: ${MY_POD_IP}:4317 - http: - endpoint: ${MY_POD_IP}:4318 - zipkin: - endpoint: ${MY_POD_IP}:9411 - jaeger: - protocols: - grpc: - endpoint: ${MY_POD_IP}:14250 - thrift_http: - endpoint: ${MY_POD_IP}:14268 - thrift_compact: - endpoint: ${MY_POD_IP}:6831 - thrift_binary: - endpoint: ${MY_POD_IP}:6832 - prometheus: - config: - scrape_configs: - - job_name: opentelemetry-collector - scrape_interval: 30s - static_configs: - - targets: - - ${MY_POD_IP}:8888 - service: - extensions: - # - zpages - # - pprof - - health_check -# - memory_ballast - telemetry: - metrics: - address: ${env:MY_POD_IP}:8888 - pipelines: - traces: - exporters: - - loadbalancing - processors: - - memory_limiter - # - spanmetrics - - batch - receivers: - - otlp - metrics: null - logs: null -tolerations: - - operator: Exists - -resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 1 - memory: 2G - -ports: - jaeger-binary: - enabled: true - containerPort: 6832 - servicePort: 6832 - hostPort: 6832 - protocol: TCP - # In order to enable podMonitor, following part must be enabled in order to expose the required port: - # metrics: - # enabled: true - -# podMonitor: -# enabled: true - -# prometheusRule: -# enabled: true -# defaultRules: -# enabled: true - -# Annotations to be added to pod -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-dsgpu-contour-ext - operator: NotIn - values: - - dedicated - - key: p-dsgpu-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-dsgpu-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-dsgpu-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-dsgpu-default-prd-ase1 - - key: cloud.google.com/compute-class - operator: NotIn - values: - - n4d-highcpu-64-vmagent-compute-class - - n4-highcpu-16-vminsert-compute-class - - c3-highcpu-44-vmselect-compute-class - - n4d-highmem-48-vmstorage-compute-class \ No newline at end of file diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index 86578f3..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,134 +0,0 @@ -fullnameOverride: opentelemetry-dsgpu-prd - -mode: daemonset - -# priorityClassName: "system-node-critical" -# Removing since dataengg pods are going into pending state -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-dsgpu-contour-ext - operator: NotIn - values: - - dedicated - - key: p-dsgpu-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-dsgpu-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-dsgpu-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-dsgpu-default-prd-ase1 - - key: cloud.google.com/compute-class - operator: NotIn - values: - - n4d-highcpu-64-vmagent-compute-class - - n4-highcpu-16-vminsert-compute-class - - c3-highcpu-44-vmselect-compute-class - - n4d-highmem-48-vmstorage-compute-class - - c3-highcpu-22-contour-internal-1-compute-class - - c4d-highcpu-16-contour-internal-1-compute-class - - c3-highcpu-22-contour-internal-0-compute-class - - c4d-highcpu-16-contour-internal-0-compute-class - -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 100m - memory: 124Mi - limits: - cpu: 200m - memory: 256Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "dsgpu" - team: "sre" - service: "opentelemetry-dsgpu-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/paused-container/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/paused-container/custom-values.yaml deleted file mode 100644 index df5354c..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/paused-container/custom-values.yaml +++ /dev/null @@ -1,82 +0,0 @@ -# Default values for hello-world. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -priorityClass: - name: "low-priority" - -image: - repository: nginx - tag: "1.14.2" - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - -nameOverride: "" -fullnameOverride: "paused-container-dsgpu-prd" - -extraLabels: - team: "devops" - bu: "dsgpu" - env: "prd" - service: "paused-container-dsgpu-prd" - priority: "p3" - type: "tools" - -topologySpreadConstraints: [] - # - labelSelector: - # matchLabels: - # dedicated: vminsert - # maxSkew: 1 - # topologyKey: topology.kubernetes.io/hostname - # whenUnsatisfiable: DoNotSchedule - -affinity: - podAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - - labelSelector: - matchExpressions: - - key: type - operator: In - values: - - vmselect - - vminsert - - tools - topologyKey: topology.kubernetes.io/zone - podAntiAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - - labelSelector: - matchExpressions: - - key: type - operator: In - values: - - vmselect - - vminsert - - tools - topologyKey: kubernetes.io/hostname - namespaceSelector: {} - - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" - -service: - type: ClusterIP - port: 80 - -deployments: - - nodepool: vminsert - bu: dsgpu - cpuRequest: 1 - memoryRequest: 1Gi - replicaCount: 3 - - nodepool: vmselect - bu: dsgpu - cpuRequest: 1 - memoryRequest: 1Gi - replicaCount: 3 diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index dac2b84..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,496 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus' global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric's labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job's name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-dataengg-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "dsgpu" - team: "sre" - service: "node-exporter-dsgpu-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index 862e026..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-dsgpu-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-dsgpu-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "dsgpu" - team: "sre" - service: "stackdriver-exporter-dsgpu-prd" - env: "prd" - priority: "p0" - type: "exporter" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-dsgpu-prd-0622" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,logging.googleapis.com/user,run.googleapis.com/,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime,compute.googleapis.com/instance' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - cloud.google.com/compute-class: c3-highcpu-44-vmselect-compute-class - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "c3-highcpu-44-vmselect-compute-class" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-dssre-stackdriver-prd@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/telegraf-operator/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/telegraf-operator/custom-values.yaml deleted file mode 100644 index c7298e2..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,150 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "telegraf-operator-dsgpu-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - cloud.google.com/compute-class: "c3-highcpu-44-vmselect-compute-class" - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "c3-highcpu-44-vmselect-compute-class" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml deleted file mode 100644 index d355229..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-dsgpu-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-dsgpu-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-dsgpu-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-dsgpu-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/datascience/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-dsgpu-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "dsgpu" - team: "sre" - service: "vmalert-dsgpu-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - cloud.google.com/compute-class: "c3-highcpu-44-vmselect-compute-class" - - - priorityClassName: "" - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "c3-highcpu-44-vmselect-compute-class" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/victoriametrics-agent/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index 6cd2798..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,479 +0,0 @@ -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "dsgpu-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -# -- Replica count -replicaCount: 2 - -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - -# -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables -deployment: - spec: - minReadySeconds: 1200 - progressDeadlineSeconds: 1800 - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - -image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - # -- Image tag, set to `Chart.AppVersion` by default - tag: v1.133.0 # rewrites Chart.AppVersion - # -- Variant of the image to use. - # e.g. enterprise, scratch - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - -# -- Image pull secrets -imagePullSecrets: [] - -# -- Add additional DNS entries to pods hosts file. Check [official documentation](https://kubernetes.io/docs/tasks/network/customize-hosts-file-for-pods/) -hostAliases: [] -# - ip: 192.168.1.1 -# hostNames: -# - test.example.com -# - another.example.net - -# -- Override chart name -nameOverride: "" - -# -- Override resources fullname -fullnameOverride: "vm-agent-dsgpu-prd" - -# -- Container working directory -containerWorkingDir: "/" - -rbac: - # -- Enables Role/RoleBinding creation - create: true - - # -- Role/RoleBinding annotations - annotations: {} - - # -- Role/RoleBinding labels - extraLabels: {} - - # -- If true and `rbac.enabled`, will deploy a Role/RoleBinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - - # -- additional rules for a role - extraRules: [] - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - # -- Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-dsci-sre-vmagnt-prd-mds@meesho-datascience-prd-0622.iam.gserviceaccount.com - } - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # -- mount API token to pod directly - automountToken: true - -# -- See `kubectl explain poddisruptionbudget.spec` for more or check [official documentation](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# -- Generates `remoteWrite.*` flags and config maps with value content for values, that are of type list of map. -# Each item should contain `url` param to pass validation. -remoteWrite: - - url: http://vm-insert-dsgpu-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write -# urlRelabelConfig: -# - action: keep -# source_labels: [env] -# regex: "dev" - -# -- VMAgent extra command line arguments -extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8429 - promscrape.config.strictParse: false # default is true - promscrape.maxScrapeSize: 1000000000 # default is 167772160 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - # promscrape.suppressScrapeErrors: true # default is false - - # Uncomment and specify the port if you want to support any of the protocols: - # https://docs.victoriametrics.com/victoriametrics/vmagent/#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for more details. -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth-secret - # key: password - -# -- Specify alternative source for env variables -envFrom: - [] - #- configMapRef: - # name: special-config - -# -- Extra labels for Deployment and Statefulset -extraLabels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "vm-agent-dsgpu-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra labels for Pods only -podLabels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "vm-agent-dsgpu-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra selector labels common for pod and service -selectorLabels: {} - -# -- Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# -- Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# -- Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -# -- Extra containers to run in a pod with vmagent -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -# -- Init containers for vmagent -initContainers: - [] - # - name: example - # image: example-image - -# -- Security context to be added to pod -podSecurityContext: - enabled: true - # fsGroup: 2000 - -# -- Security context to be added to pod's containers -securityContext: - enabled: true - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - # -- Enable agent service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - extraLabels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) for details - externalIPs: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8429 - # -- Target port - targetPort: http - # nodePort: 30000 - # -- Service type - type: ClusterIP - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - -ingress: - # -- Enable deployment of ingress for agent - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-agent-dsgpu-prd.meeshogcp.in - path: - - / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - - # -- Ingress controller class name - ingressClassName: "nginx-internal" - - # -- Ingress path type - pathType: Prefix - -# -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) -resources: - requests: - cpu: 58 - memory: 110Gi - -# -- Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) -nodeSelector: - cloud.google.com/compute-class: "n4d-highcpu-64-vmagent-compute-class" - -# -- Node tolerations for server scheduling to nodes with taints. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "n4d-highcpu-64-vmagent-compute-class" - effect: "NoSchedule" - -# -- Pod topologySpreadConstraints -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# -- Pod affinity -affinity: {} - -# -- VMAgent [scraping configuration](https://docs.victoriametrics.com/victoriametrics/vmagent/#how-to-collect-metrics-in-prometheus-format) -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-dsgpu-prd" - -# -- Priority class to be assigned to the pod(s) -priorityClassName: "" - -# -- Enable the host network -hostNetwork: false - -serviceMonitor: - # -- Enable deployment of Service Monitor for server component. This is Prometheus operator object - enabled: false - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Service Monitor relabelings - relabelings: [] - # -- Basic auth params for Service Monitor - basicAuth: {} - # -- Service Monitor metricRelabelings - metricRelabelings: [] - # -- Service Monitor targetPort - targetPort: http - # interval: 15s - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - -# -- Empty dir configuration for a case, when persistence is disabled -emptyDir: {} - -persistentVolume: - # -- Create/use Persistent Volume Claim for server component. Empty dir if false - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Size of the volume. Should be calculated based on the logs you send and retention policy you set. - size: 10Gi - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume additional labels - extraLabels: {} - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Horizontal Pod Autoscaling. -# Note that it is not intended to be used for vmagents which perform scraping. -# In order to scale scraping vmagents check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) -horizontalPodAutoscaling: - # -- Use HPA for vmagent - enabled: false - # -- Maximum replicas for HPA to use to to scale vmagent - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale vmagent - minReplicas: 1 - # -- Metric for HPA to use to scale vmagent - metrics: [] - -# -- VMAgent scrape configuration -config: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -probe: - # -- Readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - # -- Liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - # -- Startup probe - startup: {} - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -allowedMetricsEndpoints: - - /metrics - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/victoriametrics-insert/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/victoriametrics-insert/custom-values.yaml deleted file mode 100644 index 61caffa..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/victoriametrics-insert/custom-values.yaml +++ /dev/null @@ -1,411 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: false - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- IDs of vmstorage nodes to exclude from writing - excludeStorageIDs: [] - # -- Override default `app` label name - name: vminsert - # -- VMInsert strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - # override Chart.AppVersion - tag: v1.107.0-cluster - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMInsert http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-dsgpu-prd - # -- Extra command line arguments for vminsert component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8480 - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-dsgpu-prd-0.vm-storage-dsgpu-prd.victoriametrics.svc:8400" - - "vm-storage-dsgpu-prd-1.vm-storage-dsgpu-prd.victoriametrics.svc:8400" - - "vm-storage-dsgpu-prd-2.vm-storage-dsgpu-prd.victoriametrics.svc:8400" - - "vm-storage-dsgpu-prd-3.vm-storage-dsgpu-prd.victoriametrics.svc:8400" - - "vm-storage-dsgpu-prd-4.vm-storage-dsgpu-prd.victoriametrics.svc:8400" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "vm-insert-dsgpu-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Pod's additional labels - podLabels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "vm-insert-dsgpu-prd" - env: "prd" - priority: "p0" - type: "vminsert" - terminationGracePeriodSeconds: 30 - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - - # -- Readiness & Liveness probes - probe: - # -- VMInsert readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert startup probe - startup: {} - - # -- Relabel configuration - relabel: - enabled: false - config: [] - # -- Use existing configmap if specified - # otherwise .config values will be used. Relabel config **should** reside under `relabel.yml` key - configMap: "" - - # Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 7 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vminsert - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vminsert - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "n4-highcpu-16-vminsert-compute-class" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#nodeselector) - nodeSelector: - cloud.google.com/compute-class: "n4-highcpu-16-vminsert-compute-class" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 10 - memory: 10Gi - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - service: - # -- Create VMInsert service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here]( https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) for details - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-insert-dsgpu-prd.meeshogcp.in - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - - # -- Ingress controller class name - ingressClassName: nginx-internal - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for insert component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/insert" .Values.vminsert }}' - - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vminsert component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/victoriametrics-select/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/victoriametrics-select/custom-values.yaml deleted file mode 100644 index ff39417..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/victoriametrics-select/custom-values.yaml +++ /dev/null @@ -1,444 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - splitService: true - externalService: - enabled: true - name: vmselect-dsgpu-prd-proxy - # -- Override default `app` label name - clusternativeService: - enabled: false - name: "" - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # override Chart.AppVersion - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMSelect http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-dsgpu-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - # -- Extra command line arguments for vmselect component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8481 - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-dsgpu-prd-0.vm-storage-dsgpu-prd.victoriametrics.svc:8401" - - "vm-storage-dsgpu-prd-1.vm-storage-dsgpu-prd.victoriametrics.svc:8401" - - "vm-storage-dsgpu-prd-2.vm-storage-dsgpu-prd.victoriametrics.svc:8401" - - "vm-storage-dsgpu-prd-3.vm-storage-dsgpu-prd.victoriametrics.svc:8401" - - "vm-storage-dsgpu-prd-4.vm-storage-dsgpu-prd.victoriametrics.svc:8401" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "vm-select-dsgpu-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # -- Pod's additional labels - podLabels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "vm-select-dsgpu-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Readiness & Liveness probes - probe: - # -- VMSelect readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 10 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vmselect - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vmselect - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "c3-highcpu-44-vmselect-compute-class" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - cloud.google.com/compute-class: "c3-highcpu-44-vmselect-compute-class" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 38 - memory: 70Gi - - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Create VMSelect service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: false - - # -- Ingress annotations - annotations: {} - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-select-dsgpu.prd.meesho.int - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # -- Ingress controller class name - ingressClassName: contour-internal-0 - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for select component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/select" .Values.vmselect }}' - - # -- vmselect mode: deployment, daemonSet - mode: deployment - - # -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables - deployment: - spec: - # -- VMSelect strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - - # -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables - statefulSet: - enabled: false - spec: - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume extra labels - extraLabels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmselect component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: false - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false diff --git a/helm-overrides/k8s-dsgpu-prd-ase1/victoriametrics-storage/custom-values.yaml b/helm-overrides/k8s-dsgpu-prd-ase1/victoriametrics-storage/custom-values.yaml deleted file mode 100644 index dc95ffe..0000000 --- a/helm-overrides/k8s-dsgpu-prd-ase1/victoriametrics-storage/custom-values.yaml +++ /dev/null @@ -1,363 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - enabled: false - -vminsert: - enabled: false - -vmauth: - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- Override default `app` label name - name: vmstorage - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Variant of the image to use. e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMStorage http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-dsgpu-prd - - # -- Additional environment variables (ex.: secret tokens, flags). Check https://docs.victoriametrics.com/victoriametrics/#environment-variables for details - env: [] - # -- Specify alternative source for env variables - envFrom: [] - # -- Data retention period. Possible units character: h(ours), d(ays), w(eeks), y(ears), if no unit character specified - month. The minimum retention period is 24h. - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - httpListenAddr: :8482 - loggerTimezone: "Asia/Kolkata" - dedup.minScrapeInterval: 60s - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "n4d-highmem-48-vmstorage-compute-class" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - cloud.google.com/compute-class: "n4d-highmem-48-vmstorage-compute-class" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - # -- Use an alternate scheduler, e.g. "stork". Check https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ for details - # schedulerName: - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Override Persistent Volume Claim name - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are https://kubernetes.io/docs/concepts/storage/persistent-volumes/ - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume extra labels - extraLabels: {} - # -- Storage class name. Will be empty if not set - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume - size: 8134Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "vm-storage-dsgpu-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - # -- Pod's additional labels - podLabels: - bu: "dsgpu" - team: "dsgpu-sre" - service: "vm-storage-dsgpu-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Details are https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 42 - memory: 330Gi - - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: - enabled: false - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - podSecurityContext: - enabled: false - service: - enabled: true - # -- Service annotations - annotations: {} - # -- Service ClusterIP - clusterIP: None - # -- Service type - type: ClusterIP - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmselect - vmselectPort: 8401 - # -- Extra service ports - extraPorts: [] - # -- Health check node port for a service. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilyPolicy: "" - # -- List of service IP families. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilies: [] - # -- Traffic Distribution. Check https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution - trafficDistribution: "" - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - minReadySeconds: 5 - # -- Readiness probes - probe: - # -- VMStorage readiness probe - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMStorage startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmstorage component - enabled: false - # -- Maximum replicas for HPA to use to to scale the vmstorage component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmstorage component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmstorage component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: - scaleDown: - selectPolicy: Disabled - - vmbackupmanager: - # -- Enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-farmiso-prd-ase1/README.md b/helm-overrides/k8s-farmiso-prd-ase1/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/k8s-farmiso-prd-ase1/alloy/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/alloy/custom-values.yaml deleted file mode 100644 index 5dad24a..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/alloy/custom-values.yaml +++ /dev/null @@ -1,57 +0,0 @@ -fullnameOverride: "alloy-farmiso-prd" - -alloy: - configMap: - configFile: farmiso.alloy - - clustering: - enabled: true - - extraPorts: - - name: "otlp-grpc" - port: 4317 - targetPort: 4317 - protocol: "TCP" - - name: "otlp-http" - port: 4318 - targetPort: 4318 - protocol: "TCP" - - resources: - requests: - cpu: 3 - memory: 12Gi - -contour: - enabled: true - instances: [ - "contour-internal-0", - "contour-external", - ] - -configReloader: - enabled: true - -serviceAccount: - annotations: { - iam.gke.io/gcp-service-account: sa-farm-sre-grafna-obs-stk-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - } - -controller: - # Must be one of 'daemonset', 'deployment', or 'statefulset'. - type: 'deployment' - - nodeSelector: - dedicated: "alloy" - tolerations: - - key: "dedicated" - operator: "Equal" - value: "alloy" - effect: "NoSchedule" - - autoscaling: - enabled: true - minReplicas: 1 - maxReplicas: 50 - targetCPUUtilizationPercentage: 80 - targetMemoryUtilizationPercentage: 80 \ No newline at end of file diff --git a/helm-overrides/k8s-farmiso-prd-ase1/aurva-dataplane/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/aurva-dataplane/custom-values.yaml deleted file mode 100644 index fb2a0b3..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/aurva-dataplane/custom-values.yaml +++ /dev/null @@ -1,673 +0,0 @@ -## Aurva Data Plane -## Ref: https://github.com/aurva-io/aurva-charts.git - -postgresql: - enabled: true - fullnameOverride: "aurva-dataplane-database" - volumePermissions: - ## @param volumePermissions.enabled Enable init container that changes the owner and group of the persistent volume - ## - enabled: true - global: - storageClass: pd-standard-retain - postgresql: - auth: - postgresPassword: "aurva" - database: "controller" - # Add toleration to make sure where this postgres db pod should reside (Applicable for production workloads): For more detail ref: https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/ - primary: - extendedConfiguration: | - max_connections = 300 - - #PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: farmiso-devops - - # -- Select nodes to deploy which matches the following labels - nodeSelector: ##PLACEHOLDER## - dedicated: farmiso-devops - -# -- Provide a name in place of `aurva` -# namespaceOverride: aurva-dataplane - -########################################################## -# Global Configs -########################################################## -global: - aurva_controller: - enabled: true - aurva_fastdet: - enabled: false - aurva_pii_analyzer: - enabled: true - aurva_ocr: - enabled: true - aurva_collector: - enabled: true - - deploymentAnnotations: {} - - priorityClassName: "" - -########################################################## -# Aurva Controller -########################################################## -aurva_controller: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "farmiso" - team: "farmiso-devops" - service: "aurva-farmiso-prd" - env: "prd" - priority: "p0" - type: "aurva_controller" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 10 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - image: - # Image of the app container - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-controller/aurva-controller - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - - type: secret - name: aurva-controller-secrets - # - type: configmap - # name: proxy-datasource-config - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 2 - requests: - memory: 1Gi - cpu: 1 - - aurvaFastdet: - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-fastdet/aurva-fastdet - tag: "v2.30.13" - pullPolicy: IfNotPresent - envFrom: [] - env: [] - resources: - limits: - cpu: 0.5 - memory: 512Mi - requests: - cpu: 0.5 - memory: 512Mi - - nodeSelector: ##PLACEHOLDER## - dedicated: farmiso-devops - - # -- Taint tolerations for nodes - ##PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: farmiso-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-controller-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - config: - #variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - COMMAND_URL: "command.aurva-prd.meeshogcp.in:80" - DEPLOYMENT_TYPE: "kubernetes" - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - FLUSHER_WORKER_POOL_SIZE: "1000" - FLUSHER_BATCH_SIZE: "30000" - UNIQUENESS_IDENTIFIER: "k8s-farmiso-prd-ase1" #Recommendation: should be equal to cluster name - PROVIDER_ACCOUNT_ID: "meesho-farmiso-prd-0622" # GCP PROJECT ID (not Number) - REGION: "asia-southeast1" #eg: asia-south1 - ENVIRONMENT: "prod" - OCR_ENABLED: "true" - MONITORING_ENABLED: "false" - HYBRID_ONLY_MODE: "false" - FORCE_TLS: "false" - FASTDET_FLAG : "true" - AADHAAR_ENHANCER: "0" - WORKSPACE_EVENT_TRACKING_ENABLED: "false" - ACCESS_IQ_ENABLED: "true" - GRPC_ENFORCE_ALPN_ENABLED: "false" - ENABLE_FASTDET: "true" - HEARTBEAT_INTERVAL: "5m" - #pii data - PII_BUCKET_NAME: "gcs-infra-devop-aurva-farmiso-prd" - PII_LOG_BUCKET_REGION: "asia-southeast1" - PII_LOG_CRON: "*/5 * * * *" - ENABLE_PII_LOG: "true" - PII_EVIDENCE_UPLOAD_MAX_BLOCKING_TASKS: "50000" - PII_EVIDENCE_UPLOAD_MAX_CONCURRENT_TASKS: "500" - PII_EVIDENCE_MAX_CACHE_WEIGHT: "100" - QUOTA_CLEANUP_CRON: "0 0 * * *" - ENABLE_PII_QUOTA: "true" - MAX_PII_EVIDENCES_PER_KEY_PER_WINDOW: "3" - PII_QUOTA_SYNC_CRON: "*/2 * * * *" - QUOTA_WINDOW_HOURS: "24" - #constants - SKIP_NAMESPACES: "contour-internal-1-farmiso-prd,contour-internal-0-farmiso-prd,contour-external-farmiso-prd" - CLOUD_PROVIDER: "gcp" - LOG_ENV: "production" - RDS_SCANNER_AVAILABILITY : "false" - REDSHIFT_SCANNER_AVAILABILITY : "false" - S3_SCANNER_AVAILABILITY : "false" - DYNAMO_SCANNER_AVAILABILITY: "false" - DOCDB_SCANNER_AVAILABILITY: "false" - OPENSEARCH_SCANNER_AVAILABILITY: "false" - CLOUDSQL_SCANNER_AVAILABILITY: "true" - BIGQUERY_SCANNER_AVAILABILITY: "true" - AWS_SNAPSHOT_SCANNER_AVAILABILITY: "false" - CLOUDSTORAGE_SCANNER_AVAILABILITY: "true" - KEYSPACES_SCANNER_AVAILABILITY: "false" - ALLOYDB_SCANNER_AVAILABILITY: "true" - BIGTABLE_SCANNER_AVAILABILITY: "true" - GCP_BACKUP_AVAILABILITY: "true" - EGRESS_MODE_ONLY: "false" - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - SCAN_UUID_ENABLED: "true" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-controller-sa - # -- Annotations applied to created service account - annotations: - iam.gke.io/gcp-service-account: sa-farmiso-prd-aurva-contr@meesho-farmiso-prd-0622.iam.gserviceaccount.com -# eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 15 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 - - -########################################################## -# Aurva OCR -########################################################## -aurva_ocr: - # -- Additional labels for aurva-controller - additionalLabels: - bu: "farmiso" - team: "farmiso-devops" - service: "aurva-farmiso-prd" - env: "prd" - priority: "p0" - type: "aurva_ocr" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 1 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-ocr/aurva-ocr - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-ocr: - type: secret - name: aurva-ocr-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 1 - requests: - memory: 2Gi - cpu: 1 - - # -- Select nodes to deploy which matches the following labels - - nodeSelector: ##PLACEHOLDER## - dedicated: farmiso-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: farmiso-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-ocr-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - OCR_TIME_LIMIT: "1" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-farmiso-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-ocr-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # iam.gke.io/gcp-service-account: service-account@gcp.iam.gserviceaccount.com - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva Collector -########################################################## -aurva_collector: - - # -- Additional labels for aurva-analyzer - additionalLabels: - bu: "farmiso" - team: "farmiso-devops" - service: "aurva-farmiso-prd" - env: "prd" - priority: "p0" - type: "aurva_collector" - - - # -- Annotations on aurva-analyzer - annotations: {} -# "key": "value" - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-collector/aurva-collector - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-controller: - type: secret - name: aurva-collector-secrets - - resources: - limits: - cpu: 800m - memory: 800Mi - requests: - cpu: 200m - memory: 512Mi - - podSecurityContext: {} - - securityContext: - privileged: true - capabilities: - add: - # For kernel v5.8 and above we don't need CAP_SYS_ADMIN or CAP_SYS_RESOURCE - # we just need CAP_BPF and CAP_PERFMON. This has been tested on our EKS node - # which is on kernel v5.10.x - # When SSL Tracing is required we need CAP_SYS_ADMIN and CAP_SYS_PTRACE - # on top of the previous capabilities - # So finally these are the 4 possible combinations for capabilies - # 1. Newer Kernels without SSL - # - BPF - # - PERFMON - # 2. Newer Kernels with SSL - - SYS_ADMIN - - SYS_PTRACE - # 3. Older Kernels without SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # 4. Older Kernels with SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # - SYS_PTRACE - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - volumes: - - name: debugfs - mountPath: /sys/kernel/debug - hostPath: /sys/kernel/debug - - name: vmlinux - mountPath: /sys/kernel/btf/vmlinux - hostPath: /sys/kernel/btf/vmlinux - - name: procfs - mountPath: /host/proc - hostPath: /proc - - name: bpffs - mountPath: /sys/fs/bpf - hostPath: /sys/fs/bpf - - # -- Taint tolerations for nodes - tolerations: - # - effect: NoSchedule - # key: dedicated - # operator: Equal - # value: megatetra - - operator: Exists - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external - - alloy - - preprod-spot-16 - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-collector-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - # variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-farmiso-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - TRACE_INTERNAL_SVC: "true" - TRACE_INTERNAL_SVC_HTTP: "true" - INTERNAL_SVC_SAMPLE_INTERVAL: "10m" - LOGS_TTL: "1h" - TRACE_HTTP2: "true" - TRACE_SSL: "false" - TRACE_PSQL: "false" - TRACE_SQLSERVER: "false" - TRACE_MYSQL: "false" - TRACE_EGRESS: "true" - TRACE_GO_TLS: "false" - TRACE_ML_SERVICES: "false" - # constants - LOG_ENV: production - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - MONITORING_ENABLED: "false" - ENABLE_INGRESS_INFORMER: "false" - ENABLE_SERVICE_INFORMER: "false" - ENABLE_ISTIO_INFORMER: "false" - EXCLUDED_PII_REGEX_TYPES: "ip_address,us_bank_number,us_driver_license,us_itin,us_passport,us_routing,us_mbi,ssn" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-collector-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva PII Analyzer -########################################################## -aurva_pii_analyzer: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "farmiso" - team: "farmiso-devops" - service: "aurva-farmiso-prd" - env: "prd" - priority: "p0" - type: "aurva_pii_analyzer" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 3 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - ##PLACEHOLDER## - nodeSelector: - dedicated: farmiso-devops - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-piianalyzer/aurva-piianalyzer - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-pii-analyzer: - type: secret - name: aurva-pii-analyzer-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 4Gi - cpu: 4 - requests: - memory: 2Gi - cpu: 2 - - nodeSelector: ##PLACEHOLDER## - dedicated: farmiso-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: farmiso-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-pii-analyzer-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - SCHEDULER_TIME: "1" - SUPPORTED_REGION: "US" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-farmiso-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-pii-analyzer-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 3 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 diff --git a/helm-overrides/k8s-farmiso-prd-ase1/cert-manager/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/cert-manager/custom-values.yaml deleted file mode 100644 index 0c3de89..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/cert-manager/custom-values.yaml +++ /dev/null @@ -1,129 +0,0 @@ -global: - logLevel: 2 - rbac: - create: true - priorityClassName: "high-priority" -installCRDs: false - -crds: - enabled: true - keep: true - -# Cert-manager Controller -replicaCount: 1 -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-controller - tag: v1.20.1 - pullPolicy: IfNotPresent - -nodeSelector: - dedicated: farmiso-devops - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "farmiso-devops" - effect: "NoSchedule" - -resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 500m - memory: 512Mi - -# Webhook Configuration -webhook: - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-webhook - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: farmiso-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "farmiso-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# CA Injector Configuration -cainjector: - enabled: true - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-cainjector - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: farmiso-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "farmiso-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# ACME Solver Configuration -acmesolver: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-acmesolver - tag: v1.20.1 - pullPolicy: IfNotPresent - -# Startup API Check -startupapicheck: - enabled: true - timeout: 1m - backoffLimit: 4 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-startupapicheck - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: farmiso-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "farmiso-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - -# Prometheus Monitoring -prometheus: - enabled: true - servicemonitor: - enabled: false - interval: 60s - scrapeTimeout: 30s - labels: - prometheus: cert-manager diff --git a/helm-overrides/k8s-farmiso-prd-ase1/conntrack-adjuster/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 02c4a82..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,17 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - max: 2097152 - hashsize: 524288 - sleepInterval: 30 - -nodeAffinity: - enabled: true - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: In - values: - - "contour-external" - - "contour-internal-0" diff --git a/helm-overrides/k8s-farmiso-prd-ase1/contour-cert-checker/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/contour-cert-checker/custom-values.yaml deleted file mode 100644 index 0e22981..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=k8s-farmiso-prd-ase1"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - dedicated: farmiso-devops - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: farmiso-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/k8s-farmiso-prd-ase1/contour-external/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/contour-external/custom-values.yaml deleted file mode 100644 index 6975c58..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/contour-external/custom-values.yaml +++ /dev/null @@ -1,101 +0,0 @@ -configInline: - enableExternalNameService: true - network: - num-trusted-hops: 1 - timeouts: - connection-idle-timeout: 305s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled - tracing: - includePodDetail: true - extensionService: observability/contour-external-extension-service - serviceName: contour-external-farmiso-prd - customTags: - - tagName: header-tag - requestHeaderName: X-Custom-Header -contour: - enabled: true - replicaCount: 2 - podLabels: - bu: farmiso - team: farmiso-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external - nodeSelector: - dedicated: contour-external - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: farmiso - team: farmiso-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external - nodeSelector: - dedicated: contour-external - logLevel: error - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "40" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-farmiso-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false - -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend-new" diff --git a/helm-overrides/k8s-farmiso-prd-ase1/contour-internal-0/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/contour-internal-0/custom-values.yaml deleted file mode 100644 index b16642b..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,104 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled - tracing: - includePodDetail: true - extensionService: observability/contour-internal-0-extension-service - serviceName: contour-internal-0-farmiso-prd - customTags: - - tagName: header-tag - requestHeaderName: X-Custom-Header -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 2 - podLabels: - bu: farmiso - team: farmiso-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: farmiso - team: farmiso-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-farmiso-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false - -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend-new" - diff --git a/helm-overrides/k8s-farmiso-prd-ase1/contour-internal-intra-0/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index fdb424a..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,101 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled - tracing: - includePodDetail: true - extensionService: observability/contour-internal-0-extension-service - serviceName: contour-internal-0-farmiso-prd - customTags: - - tagName: header-tag - requestHeaderName: X-Custom-Header -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 2 - podLabels: - bu: farmiso - team: farmiso-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: farmiso - team: farmiso-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false - -certManager: - externalSecret: - enabled: true - secretStoreRef: - name: "vault-backend-new" diff --git a/helm-overrides/k8s-farmiso-prd-ase1/coredns/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/coredns/custom-values.yaml deleted file mode 100644 index 625309c..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/coredns/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -replicaCount: 5 -communicationType: "intra" - -labels: - bu: farmiso - team: farmiso-devops - env: prd - -clusterIP: 10.137.4.2 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "farmiso-devops" - effect: "NoSchedule" - -nodeSelector: - dedicated: farmiso-devops - kubernetes.io/os: linux \ No newline at end of file diff --git a/helm-overrides/k8s-farmiso-prd-ase1/external-secrets/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/external-secrets/custom-values.yaml deleted file mode 100644 index 36339c9..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/external-secrets/custom-values.yaml +++ /dev/null @@ -1,61 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Annotations to add to Pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: farmiso-devops - nodeSelector: - dedicated: farmiso-devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: farmiso-devops - nodeSelector: - dedicated: farmiso-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: farmiso-devops - nodeSelector: - dedicated: farmiso-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" \ No newline at end of file diff --git a/helm-overrides/k8s-farmiso-prd-ase1/flagger/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/flagger/custom-values.yaml deleted file mode 100644 index a7c20e6..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/flagger/custom-values.yaml +++ /dev/null @@ -1,62 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "512Mi" - cpu: "1000m" - requests: - memory: "256Mi" - cpu: "100m" - -nodeSelector: - dedicated: farmiso-devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: farmiso-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: farmiso-devops - bu: infra - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-farmiso-rollout-service.prd-farmiso-rollout-service.svc.cluster.local" - -rolloutService: - enabled: true - apps: \ No newline at end of file diff --git a/helm-overrides/k8s-farmiso-prd-ase1/fluentd/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/fluentd/custom-values.yaml deleted file mode 100644 index e2c24d9..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/fluentd/custom-values.yaml +++ /dev/null @@ -1,812 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-farm-fmsre-fluentd-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: farmiso - team: farmiso-sre - type: fluentd - service: fluentd-farmiso-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: farmiso - team: farmiso-sre - type: fluentd - service: fluentd-farmiso-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend-new - path: meesho/prd/cntr/devop/coralogix-keys - -envFrom: -# - secretRef: -# name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: - - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - from_encoding "utf-8" - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-farmiso-prd-ase1/keda/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/keda/custom-values.yaml deleted file mode 100644 index bac2309..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/keda/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: farmiso-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "farmiso-devops" - effect: "NoSchedule" - podLabels: - bu: "farmiso" - team: "farmiso-devops" - metricsAdapter: - bu: "farmiso" - team: "farmiso-devops" - resources: - webhooks: - limits: - cpu: 50m - memory: 100Mi - requests: - cpu: 10m - memory: 20Mi \ No newline at end of file diff --git a/helm-overrides/k8s-farmiso-prd-ase1/kube-dns/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/kube-dns/custom-values.yaml deleted file mode 100644 index 835a478..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.137.4.2"],"prd.mrouter.int.svc.cluster.local":["10.137.4.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.137.4.2"]} diff --git a/helm-overrides/k8s-farmiso-prd-ase1/kube-events/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/kube-events/custom-values.yaml deleted file mode 100644 index e73e8c3..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-farmiso-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - dedicated: farmiso-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: farmiso-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: farmiso-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: farmiso-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: farmiso-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: farmiso-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-farmiso-prd-ase1/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 316fa5a..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: "" - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-farmiso-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-farmiso-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "kube-state-metrics-farmiso-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "vmselect-mds" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 10m - memory: 50Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-farmiso-prd-ase1/kubectl-mcp-server/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/kubectl-mcp-server/custom-values.yaml deleted file mode 100644 index a420337..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/kubectl-mcp-server/custom-values.yaml +++ /dev/null @@ -1,119 +0,0 @@ -fullnameOverride: "kubectl-mcp-server" - -replicas: 1 - -image: - # In-house hardened build (helm-templates/kubectl-mcp-server/docker/) — - # NOT the upstream Docker Hub image. 266→18 HIGH/CRIT vulns, non-root, - # multi-stage slim base, kubectl v1.33.12 / helm v3.21.0. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/prd/devop/kubectl-mcp-server - tag: "v2" - pullPolicy: IfNotPresent - -labels: - bu: infra - team: devops - service: kubectl-mcp-server - env: prd - -serviceAccount: - create: true - annotations: {} - -rbac: - create: true - -podAnnotations: {} - -podSecurityContext: {} - -securityContext: - runAsNonRoot: true - runAsUser: 1000 - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -priorityClassName: "" - -nodeSelector: - dedicated: "farmiso-devops" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "farmiso-devops" - effect: NoSchedule - -affinity: {} - -resources: - requests: - cpu: 250m - memory: 512Mi - limits: - cpu: 1000m - memory: 1Gi - -mcp: - mode: single - transport: http - host: "0.0.0.0" - port: 8000 - -auth: - allowAnonymous: false - -externalSecrets: - enabled: true - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/prd/cntr/devop/kubectl-mcp-server-farmiso" - -# tcpSocket on purpose — streamable-http transport exposes only /mcp, -# there is no /health route (see chart values.yaml note). -livenessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - -readinessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - -service: - port: 8000 - type: ClusterIP - -# Contour standard (2.0.0-style): renders HTTPProxy parent/child + -intra -# instead of a networking.k8s.io Ingress. -createContourGateway: true - -ingress: - enabled: true - # farmiso prd exposes the -0 contour internal classes (no -1); the chart - # auto-derives contour-internal-intra-0. Verified on k8s-farmiso-prd-ase1. - ingressClassName: contour-internal-0 - servicePortNumber: 8000 - hosts: - - host: kubectl-mcp-server-farmiso.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 diff --git a/helm-overrides/k8s-farmiso-prd-ase1/kyverno/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/kyverno/custom-values.yaml deleted file mode 100644 index b349ab7..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/kyverno/custom-values.yaml +++ /dev/null @@ -1,2240 +0,0 @@ -global: - - # -- Internal settings used with `helm template` to generate install manifest - # @ignored - templating: - enabled: false - debug: false - version: ~ - - image: - # -- (string) Global value that allows to set a single image registry across all deployments. - # When set, it will override any values set under `.image.registry` across the chart. - registry: ~ - # -- (list) Global list of Image pull secrets - # When set, it will override any values set under `imagePullSecrets` under different components across the chart. - imagePullSecrets: [] - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - caCertificates: - # -- Global CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - # Individual controller values will override this global value - data: ~ - - # -- Global value to set single volume to be mounted for CA certificates for all deployments. - # Not used when `.Values.global.caCertificates.data` is defined - # Individual controller values will override this global value - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Additional container environment variables to apply to all containers and init containers - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Global node labels for pod assignment. Non-global values will override the global value. - nodeSelector: - dedicated: farmiso-kyverno - - # -- Global List of node taints to tolerate. Non-global values will override the global value. - tolerations: - - key: "dedicated" - operator: "Equal" - value: "farmiso-kyverno" - effect: "NoSchedule" - -# -- (string) Override the name of the chart -nameOverride: ~ - -# -- (string) Override the expanded name of the chart -fullnameOverride: ~ - -# -- (string) Override the namespace the chart deploys to -namespaceOverride: ~ - -upgrade: - # -- Upgrading from v2 to v3 is not allowed by default, set this to true once changes have been reviewed. - fromV2: false - -apiVersionOverride: - # -- (string) Override api version used to create `PodDisruptionBudget`` resources. - # When not specified the chart will check if `policy/v1/PodDisruptionBudget` is available to - # determine the api version automatically. - podDisruptionBudget: ~ - -rbac: - roles: - # -- Aggregate ClusterRoles to Kubernetes default user-facing roles. For more information, see [User-facing roles](https://kubernetes.io/docs/reference/access-authn-authz/rbac/#user-facing-roles) - aggregate: - admin: true - view: true - -# Use openreports.io as the API group for reporting -openreports: - # -- Enable OpenReports feature in controllers - enabled: false - # -- Whether to install CRDs from the upstream OpenReports chart. Setting this to true requires enabled to also be true. - installCrds: false - -# CRDs configuration -crds: - - # -- Whether to have Helm install the Kyverno CRDs, if the CRDs are not installed by Helm, they must be added before policies can be created - install: true - - reportsServer: - # -- Kyverno reports-server is used in your cluster - enabled: false - - groups: - - # -- Install CRDs in group `kyverno.io` - kyverno: - cleanuppolicies: true - clustercleanuppolicies: true - clusterpolicies: true - globalcontextentries: true - policies: true - policyexceptions: true - updaterequests: true - - # -- Install CRDs in group `policies.kyverno.io` - policies: - validatingpolicies: true - policyexceptions: true - imagevalidatingpolicies: true - namespacedimagevalidatingpolicies: true - mutatingpolicies: true - generatingpolicies: true - deletingpolicies: true - namespaceddeletingpolicies: true - namespacedvalidatingpolicies: true - - # -- Install CRDs in group `reports.kyverno.io` - reports: - clusterephemeralreports: true - ephemeralreports: true - - # -- Install CRDs in group `wgpolicyk8s.io` - wgpolicyk8s: - clusterpolicyreports: false ##lk - policyreports: false ##lk - - # -- Additional CRDs annotations (Replace=true avoids 262144-byte annotation limit on large CRDs) - annotations: - argocd.argoproj.io/sync-options: Replace=true - # strategy.spinnaker.io/replace: 'true' - - # -- Additional CRDs labels - customLabels: {} - - migration: - - # -- Enable CRDs migration using helm post upgrade hook - enabled: true - - # -- Resources to migrate - resources: - - cleanuppolicies.kyverno.io - - clustercleanuppolicies.kyverno.io - - clusterpolicies.kyverno.io - - globalcontextentries.kyverno.io - - policies.kyverno.io - - policyexceptions.kyverno.io - - updaterequests.kyverno.io - - deletingpolicies.policies.kyverno.io - - generatingpolicies.policies.kyverno.io - - imagevalidatingpolicies.policies.kyverno.io - - namespacedimagevalidatingpolicies.policies.kyverno.io - - mutatingpolicies.policies.kyverno.io - - namespaceddeletingpolicies.policies.kyverno.io - - namespacedvalidatingpolicies.policies.kyverno.io - - policyexceptions.policies.kyverno.io - - validatingpolicies.policies.kyverno.io - - image: - # -- (string) Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- (string) Image repository - repository: kyverno/kyverno-cli - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- (string) Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podResources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -# Configuration -config: - - # -- Create the configmap. - create: true - - # -- Preserve the configmap settings during upgrade. - preserve: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - # -- Enable registry mutation for container images. Enabled by default. - enableDefaultRegistryMutation: true - - # -- The registry hostname used for the image mutation. - defaultRegistry: docker.io - - # -- Exclude groups - excludeGroups: - - system:nodes - - # -- Exclude usernames - excludeUsernames: [] - # - '!system:kube-scheduler' - - # -- Exclude roles - excludeRoles: [] - - # -- Exclude roles - excludeClusterRoles: [] - - # -- Generate success events. - generateSuccessEvents: false - - # -- Resource types to be skipped by the Kyverno policy engine. - # Make sure to surround each entry in quotes so that it doesn't get parsed as a nested YAML list. - # These are joined together without spaces, run through `tpl`, and the result is set in the config map. - # @default -- See [values.yaml](values.yaml) - resourceFilters: - - '[Event,*,*]' - # - '[*/*,kube-system,*]' - - '[*/*,kube-public,*]' - - '[*/*,kube-node-lease,*]' - - '[Node,*,*]' - - '[Node/?*,*,*]' - - '[APIService,*,*]' - - '[APIService/?*,*,*]' - - '[TokenReview,*,*]' - - '[SubjectAccessReview,*,*]' - - '[SelfSubjectAccessReview,*,*]' - - '[Binding,*,*]' - - '[Pod/binding,*,*]' - - '[ReplicaSet,*,*]' - - '[ReplicaSet/?*,*,*]' - - '[EphemeralReport,*,*]' - - '[ClusterEphemeralReport,*,*]' - # exclude resources from the chart - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:additional]' - - '[ClusterRoleBinding,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.configMapName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.metricsConfigMapName" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Job,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[Job/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.admission-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.background-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - # -- Sets the threshold for the total number of UpdateRequests generated for mutateExisitng and generate policies. - updateRequestThreshold: 1000 - - # -- Defines the `namespaceSelector`/`objectSelector` in the webhook configurations. - # The Kyverno namespace is excluded if `excludeKyvernoNamespace` is `true` (default) - webhooks: - # Exclude namespaces - # namespaceSelector: lk-s2 - # matchExpressions: - # - key: kubernetes.io/metadata.name - # operator: NotIn - # values: - # - kube-systemm lk-s2 - # Exclude objects - # objectSelector: - # matchExpressions: - # - key: webhooks.kyverno.io/exclude - # operator: DoesNotExist - - # -- Defines annotations to set on webhook configurations. - webhookAnnotations: - # Example to disable admission enforcer on AKS: - # 'admissions.enforcer/disabled': 'true' lk-s3 - - # -- Defines labels to set on webhook configurations. - webhookLabels: {} - # Example to adopt webhook resources in ArgoCD: - # 'argocd.argoproj.io/instance': 'kyverno' - - # -- Defines match conditions to set on webhook configurations (requires Kubernetes 1.27+). - matchConditions: [] - - # -- Exclude Kyverno namespace - # Determines if default Kyverno namespace exclusion is enabled for webhooks and resourceFilters - excludeKyvernoNamespace: true - - # -- resourceFilter namespace exclude - # Namespaces to exclude from the default resourceFilters - resourceFiltersExcludeNamespaces: [] - - # -- resourceFilters exclude list - # Items to exclude from config.resourceFilters - resourceFiltersExclude: [] - - # -- resourceFilter namespace include - # Namespaces to include to the default resourceFilters - resourceFiltersIncludeNamespaces: [] - - # -- resourceFilters include list - # Items to include to config.resourceFilters - resourceFiltersInclude: [] - -# Metrics configuration -metricsConfig: - - # -- Create the configmap. - create: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - namespaces: - - # -- List of namespaces to capture metrics for. - include: [] - - # -- list of namespaces to NOT capture metrics for. - exclude: [] - - # -- (string) Rate at which metrics should reset so as to clean up the memory footprint of kyverno metrics, if you might be expecting high memory footprint of Kyverno's metrics. Default: 0, no refresh of metrics. WARNING: This flag is not working since Kyverno 1.8.0 - metricsRefreshInterval: ~ - # metricsRefreshInterval: 24h - - # -- (list) Configures the bucket boundaries for all Histogram metrics, changing this configuration requires restart of the kyverno admission controller - bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 15, 20, 25, 30] - - # -- (map) Configures the exposure of individual metrics, by default all metrics and all labels are exported, changing this configuration requires restart of the kyverno admission controller - metricsExposure: - kyverno_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_image_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_mutating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_generating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_admission_review_duration_seconds: - # enabled: false - disabledLabelDimensions: ["resource_namespace"] - kyverno_policy_rule_info_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_policy_results_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_admission_requests_total: - disabledLabelDimensions: ["resource_namespace"] - kyverno_cleanup_controller_deletedobjects_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - -# -- Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -imagePullSecrets: {} - # regcred: - # registry: foo.example.com - # username: foobar - # password: secret - # regcred2: - # registry: bar.example.com - # username: barbaz - # password: secret2 - -# -- Existing Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -existingImagePullSecrets: [] - # - test-registry - # - other-test-registry - -# Tests configuration -test: - # -- Sleep time before running test - sleep: 20 - - image: - # -- (string) Image registry - registry: curlimages - # -- Image repository - repository: curl - # -- Image tag - # Defaults to `latest` if omitted - tag: '8.10.1' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - # -- Security context for the test containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- Additional Pod annotations - podAnnotations: {} - - # -- List of node taints to tolerate - tolerations: [] - -# -- Additional labels -customLabels: {} - - -webhooksCleanup: - # -- Create a helm pre-delete hook to cleanup webhooks. - enabled: true - - autoDeleteWebhooks: - # -- Allow webhooks controller to delete webhooks using finalizers - enabled: false - - image: - # -- (string) Image registry - registry: registry.k8s.io - # -- Image repository - repository: kubectl - # -- Image tag - # Defaults to `latest` if omitted - tag: 'v1.32.7' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -grafana: - # -- Enable grafana dashboard creation. - enabled: false - - # -- Configmap name template. - configMapName: '{{ include "kyverno.fullname" . }}-grafana' - - # -- (string) Namespace to create the grafana dashboard configmap. - # If not set, it will be created in the same namespace where the chart is deployed. - namespace: ~ - - # -- Grafana dashboard configmap annotations. - annotations: {} - - # -- Grafana dashboard configmap labels - labels: - grafana_dashboard: "1" - - # -- create GrafanaDashboard custom resource referencing to the configMap. - # according to https://grafana-operator.github.io/grafana-operator/docs/examples/dashboard_from_configmap/readme/ - grafanaDashboard: - create: false - folder: kyverno - allowCrossNamespaceImport: true - matchLabels: - dashboards: "grafana" - -# Features configuration -features: - admissionReports: - # -- Enables the feature - enabled: true - aggregateReports: - # -- Enables the feature - enabled: true - policyReports: - # -- Enables the feature - enabled: false ##lk - validatingAdmissionPolicyReports: - # -- Enables the feature - enabled: true - mutatingAdmissionPolicyReports: - # -- Enables the feature - enabled: false - reporting: - # -- Enables the feature - validate: true - # -- Enables the feature - mutate: true - # -- Enables the feature - mutateExisting: true - # -- Enables the feature - imageVerify: true - # -- Enables the feature - generate: true - autoUpdateWebhooks: - # -- Enables the feature - enabled: true - backgroundScan: - # -- Enables the feature - enabled: true - # -- Number of background scan workers - backgroundScanWorkers: 2 - # -- Background scan interval - backgroundScanInterval: 1h - # -- Skips resource filters in background scan - skipResourceFilters: true - configMapCaching: - # -- Enables the feature - enabled: true - controllerRuntimeMetrics: - # -- Bind address for controller-runtime metrics (use "0" to disable it) - bindAddress: ":8080" - deferredLoading: - # -- Enables the feature - enabled: true - dumpPayload: - # -- Enables the feature - enabled: false - forceFailurePolicyIgnore: - # -- Enables the feature - enabled: false - generateValidatingAdmissionPolicy: - # -- Enables the feature - enabled: true - generateMutatingAdmissionPolicy: - # -- Enables the feature - enabled: false - dumpPatches: - # -- Enables the feature - enabled: false - globalContext: - # -- Maximum allowed response size from API Calls. A value of 0 bypasses checks (not recommended) - maxApiCallResponseLength: 2000000 - logging: - # -- Logging format - format: text - # -- Logging verbosity - verbosity: 2 - omitEvents: - # -- Events which should not be emitted (possible values `PolicyViolation`, `PolicyApplied`, `PolicyError`, and `PolicySkipped`) - eventTypes: - - PolicyApplied - - PolicySkipped - # - PolicyViolation - # - PolicyError - policyExceptions: - # -- Enables the feature - enabled: true - # -- Restrict policy exceptions to a single namespace - # Set to "*" to allow exceptions in all namespaces - namespace: '' - protectManagedResources: - # -- Enables the feature - enabled: false - registryClient: - # -- Allow insecure registry - allowInsecure: false - # -- Enable registry client helpers - credentialHelpers: - - default - - google - - amazon - - azure - - github - ttlController: - # -- Reconciliation interval for the label based cleanup manager - reconciliationInterval: 1m - tuf: - # -- Enables the feature - enabled: false - # -- (string) Path to Tuf root - root: ~ - # -- (string) Raw Tuf root - rootRaw: ~ - # -- (string) Tuf mirror - mirror: ~ - -# Admission controller configuration -admissionController: - autoscaling: - # -- Enable horizontal pod autoscaling - enabled: true - - # -- Minimum number of pods - minReplicas: 1 - - # -- Maximum number of pods - maxReplicas: 10 - - # -- Target CPU utilization percentage - targetCPUUtilizationPercentage: 80 - - # -- Configurable scaling behavior - behavior: {} - - # -- Overrides features defined at the root level - featuresOverride: - admissionReports: - # -- Max number of admission reports allowed in flight until the admission controller stops creating new ones - backPressureThreshold: 1000 - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- The ServiceAccount name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno admission controller activities. - # This will help ensure Kyverno stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- admissionController webhook server port - # in case you are using hostNetwork: true, you might want to change the port the webhookServer is listening to - webhookServer: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - admission-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.admissionController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - initContainer: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyvernopre - # -- (string) Image tag - # If missing, defaults to image.tag - tag: ~ - # -- (string) Image pull policy - # If missing, defaults to image.pullPolicy - pullPolicy: ~ - - resources: - # -- Pod resource limits - # limits: - # cpu: 100m - # memory: 256Mi - # -- Pod resource requests - requests: - cpu: 1000m - memory: 1Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - container: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyverno - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - resources: - # -- Pod resource limits - # limits: - # memory: 384Mi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Array of extra init containers - extraInitContainers: [] - # - name: init-container - # image: busybox - # command: ['sh', '-c', 'echo Hello'] - - # -- Array of extra containers to run alongside kyverno - extraContainers: [] - # - name: myapp-container - # image: busybox - # command: ['sh', '-c', 'echo Hello && sleep 3600'] - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Kyverno's metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Background controller configuration -backgroundController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable background controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: - - apiGroups: - - networking.k8s.io - resources: - - ingresses - - ingressclasses - - networkpolicies - verbs: - - create - - update - - patch - - delete - - apiGroups: - - rbac.authorization.k8s.io - resources: - - rolebindings - - roles - verbs: - - create - - update - - patch - - delete - - apiGroups: - - '' - resources: - - configmaps - - resourcequotas - - limitranges - verbs: - - create - - update - - patch - - delete - - apiGroups: - - resource.k8s.io - resources: - - resourceclaims - - resourceclaimtemplates - verbs: - - create - - delete - - update - - patch - - deletecollection - - apiGroups: - - autoscaling - resources: - - horizontalpodautoscalers - verbs: - - get - - list - - watch - - update - - patch - - apiGroups: - - apps - resources: - - deployments - - statefulsets - - replicasets - verbs: - - get - - list - - watch - - update - - patch - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - # - patch - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/background-controller - # -- Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - # limits: - # memory: 128Mi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - background-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.backgroundController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - # -- backgroundController server port - # in case you are using hostNetwork: true, you might want to change the port the backgroundController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Cleanup controller configuration -cleanupController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable cleanup controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - delete - # - list - # - watch - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/cleanup-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- cleanupController server port - # in case you are using hostNetwork: true, you might want to change the port the cleanupController is listening to - server: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - # limits: - # memory: 128Mi - # -- Pod resource requests - requests: - cpu: 1100m - memory: 1.2Gi - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - cleanup-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `service.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Reports controller configuration -reportsController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable reports controller. - enabled: false ##lk - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/reports-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno reports controller activities. - # This will help ensure Kyverno reports stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - memory: 128Mi - # -- Pod resource requests - requests: - cpu: 100m - memory: 64Mi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - reports-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.reportsController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- (string) Service node port. - # Only used if `type` is `NodePort`. - nodePort: ~ - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- (string) Traces receiver address - address: ~ - # -- (string) Traces receiver port - port: ~ - # -- (string) Traces receiver credentials - creds: ~ - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- (string) Otel collector endpoint - collector: ~ - # -- (string) Otel collector credentials - creds: ~ - - # -- reportsController server port - # in case you are using hostNetwork: true, you might want to change the port the reportsController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - - # -- Enable sanity check for reports CRDs - sanityChecks: true diff --git a/helm-overrides/k8s-farmiso-prd-ase1/kyverno/policies/protect-namespace.yaml b/helm-overrides/k8s-farmiso-prd-ase1/kyverno/policies/protect-namespace.yaml deleted file mode 100644 index c9df380..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/kyverno/policies/protect-namespace.yaml +++ /dev/null @@ -1,40 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: protect-critical-namespaces -spec: - validationFailureAction: Enforce - background: false - rules: - - name: deny-deletion-of-critical-namespaces - match: - any: - - resources: - kinds: - - Namespace - names: - - keda-farmiso-prd - - contour-internal-0-farmiso-prd - - contour-internal-0-farmiso-prd-intra - - contour-external-farmiso-prd - - external-secrets-farmiso-prd - - flagger-farmiso-prd - - victoriametrics - - kube-system - - monitoring - - telegraf-operator - - loadtester - - opentelemetry - - fluentd - - kube-events - - observability - - contour-cert-checker-ns - - aurva-dataplane - validate: - message: "Deletion of critical namespaces is not allowed." - deny: - conditions: - any: - - key: "{{request.operation}}" - operator: Equals - value: DELETE diff --git a/helm-overrides/k8s-farmiso-prd-ase1/kyverno/policies/restrict-replicas.yaml b/helm-overrides/k8s-farmiso-prd-ase1/kyverno/policies/restrict-replicas.yaml deleted file mode 100644 index d456985..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/kyverno/policies/restrict-replicas.yaml +++ /dev/null @@ -1,29 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: restrict-minimum-replicas-selected-ns -spec: - validationFailureAction: Enforce - background: false - rules: - - name: min-replicas-deployments-statefulsets - match: - resources: - kinds: - - Deployment - - StatefulSet - - Deployment/scale - - StatefulSet/scale - names: - - coredns - namespaces: - - kube-system - operations: - - CREATE - - UPDATE - validate: - message: "Deployments and StatefulSets must have at least 5 replicas in this namespace." - anyPattern: - # Case 1: replicas explicitly set and >= 5 - - spec: - replicas: ">=5" diff --git a/helm-overrides/k8s-farmiso-prd-ase1/loadtester/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/loadtester/custom-values.yaml deleted file mode 100644 index c09b2e9..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/loadtester/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -replicaCount: 1 - -Namespace: loadtester - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/dev/devops/flagger-loadtester - tag: 0.30.0 - pullPolicy: IfNotPresent - pullSecret: - -podLabels: - bu: farmiso - team: farmiso-shared - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - -podPriorityClassName: "" - -logLevel: info -cmd: - timeout: 1h - namespaceRegexp: "" - -nameOverride: "loadtester" -fullnameOverride: "" - -env: [] - -service: - type: ClusterIP - port: 80 - -resources: - requests: - cpu: 10m - memory: 64Mi - -volumes: [] -volumeMounts: [] - -nodeSelector: - dedicated: farmiso-devops - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "farmiso-devops" - effect: "NoSchedule" - -affinity: {} - -ingress: - slowStart: - enabled: false - enabled: true - ingressClassName: contour-internal-0 - servicePort: http - hosts: - - host: farmiso-flagger-loadtester.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / -createContourGateway: true -contourResponseTimeout: false -rbac: - # rbac.create: `true` if rbac resources should be created - create: true - # rbac.scope: `cluster` to create cluster-scope rbac resources (ClusterRole/ClusterRoleBinding) - # otherwise, namespace-scope rbac resources will be created (Role/RoleBinding) - scope: - # rbac.rules: array of rules to apply to the role. example: - # rules: - # - apiGroups: [""] - # resources: ["pods"] - # verbs: ["list", "get"] - rules: [] - -# name of an existing service account to use - if not creating rbac resources -serviceAccountName: "" - -# App Mesh virtual node settings (to be used for AppMesh v1beta1) -meshName: "" -#backends: -# - app1.namespace -# - app2.namespace - -# App Mesh virtual node settings (to be used for AppMesh v1beta2) -appmesh: - enabled: false - backends: - - podinfo - - podinfo-canary - -#Istio virtual service and gatway settings. TLS secrets should be in namespace before enbaled it. ( secret format loadtester.fullname ) -istio: - enabled: false - host: flagger-loadtester.flagger - gateway: - enabled: false - tls: - enabled: false - httpsRedirect: false - -# when enabled, it will add a security context for the loadtester pod -securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 100 - runAsGroup: 101 - -podDisruptionBudget: - enabled: false - minAvailable: 1 diff --git a/helm-overrides/k8s-farmiso-prd-ase1/opentelemetry-coralogix/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/opentelemetry-coralogix/custom-values.yaml deleted file mode 100644 index 78e6fa8..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/opentelemetry-coralogix/custom-values.yaml +++ /dev/null @@ -1,281 +0,0 @@ -global: - domain: "coralogixsg.com" - defaultApplicationName: "default" - defaultSubsystemName: "nodes" - - # Old endpoint based configuration, - # please use domain instead. - traces: - endpoint: "" - metrics: - endpoint: "" - logs: - endpoint: "" - -# set distribution to openshift for openshift clusters -distribution: "" - -# Opentelemetry collector configuration -dedicatedValue: false -schedulerName: default-scheduler - -labels: - bu: "farmiso" - team: "farmiso-sre" - service: "opentelemetry-farmiso-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: false - key: prd/admin/coralogix-keys - secretStoreRef: - name: vault-backend-new - -image: - # If you want to use the core image `otel/opentelemetry-collector`, you also need to change `command.name` value to `otelcol`. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - # repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - tag: "0.87.0" - # When digest is set to a non-empty value, images will be pulled by digest (regardless of tag value). - digest: "" -mode: daemonset -rollout: - rollingUpdate: - # When 'mode: daemonset', maxSurge cannot be used when hostPort is set for any of the ports - # maxSurge: 25% - maxUnavailable: 1 - strategy: RollingUpdate -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" -hostNetwork: true -dnsPolicy: "ClusterFirstWithHostNet" -fullnameOverride: opentelemetry-farmiso-prd - -presets: - logsCollection: - enabled: false - storeCheckpoints: false - maxRecombineLogSize: 1048576 - extraFilelogOperators: [] -# - type: recombine -# combine_field: body -# source_identifier: attributes["log.file.path"] -# is_first_entry: body matches "^(YOUR-LOGS-REGEX)"s - kubernetesAttributes: - enabled: false - hostMetrics: - enabled: false - kubeletMetrics: - enabled: false - -extraEnvs: -- name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" -- name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName -config: - extensions: {} - # zpages: - # endpoint: localhost:55679 - # pprof: - # endpoint: localhost:1777 - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 4s - tls: - insecure: true - sending_queue: - enabled: true - queue_size: 6000 - num_consumers: 200 - resolver: - dns: - hostname: opentelemetry-admin-prd.opentelemetry.svc.clusterset.local - # coralogix: - # timeout: "30s" - # private_key: "${CORALOGIX_PRIVATE_KEY}" - # domain: "{{.Values.global.domain}}" - # traces: - # endpoint: "{{ .Values.global.traces.endpoint }}" - # metrics: - # endpoint: "{{ .Values.global.metrics.endpoint }}" - # logs: - # endpoint: "{{ .Values.global.logs.endpoint }}" - # application_name_attributes: - # - "k8s.namespace.name" - # - "service.namespace" - # subsystem_name_attributes: - # - "k8s.deployment.name" - # - "k8s.statefulset.name" - # - "k8s.daemonset.name" - # - "k8s.cronjob.name" - # - "k8s.job.name" - # - "k8s.container.name" - # - "k8s.node.name" - # - "service.name" - # application_name: "{{.Values.global.defaultApplicationName }}" - # subsystem_name: "{{.Values.global.defaultSubsystemName }}" - processors: - k8sattributes: - filter: - node_from_env_var: KUBE_NODE_NAME - extract: - metadata: - - "k8s.namespace.name" - - "k8s.deployment.name" - - "k8s.statefulset.name" - - "k8s.daemonset.name" - - "k8s.cronjob.name" - - "k8s.job.name" - - "k8s.pod.name" - - "k8s.node.name" - memory_limiter: null # Will get the k8s resource limits - resourcedetection/env: - detectors: ["system","env"] - timeout: 2s - override: false - # spanmetrics: - # metrics_exporter: coralogix - # dimensions: - # - name: "k8s.deployment.name" - # - name: "k8s.statefulset.name" - # - name: "k8s.daemonset.name" - # - name: "k8s.cronjob.name" - # - name: "k8s.job.name" - # - name: "k8s.container.name" - # - name: "k8s.node.name" - # - name: "k8s.namespace.name" - receivers: - otlp: - protocols: - grpc: - endpoint: ${MY_POD_IP}:4317 - http: - endpoint: ${MY_POD_IP}:4318 - zipkin: - endpoint: ${MY_POD_IP}:9411 - jaeger: - protocols: - grpc: - endpoint: ${MY_POD_IP}:14250 - thrift_http: - endpoint: ${MY_POD_IP}:14268 - thrift_compact: - endpoint: ${MY_POD_IP}:6831 - thrift_binary: - endpoint: ${MY_POD_IP}:6832 - prometheus: - config: - scrape_configs: - - job_name: opentelemetry-collector - scrape_interval: 30s - static_configs: - - targets: - - ${MY_POD_IP}:8888 - service: - extensions: - # - zpages - # - pprof - - health_check -# - memory_ballast - telemetry: - metrics: - address: ${env:MY_POD_IP}:8888 - pipelines: - traces: - exporters: - - loadbalancing - processors: - - memory_limiter - # - spanmetrics - - batch - receivers: - - otlp - metrics: null - logs: null -tolerations: - - operator: Exists - -resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 1 - memory: 2G - -ports: - jaeger-binary: - enabled: true - containerPort: 6832 - servicePort: 6832 - hostPort: 6832 - protocol: TCP - # In order to enable podMonitor, following part must be enabled in order to expose the required port: - # metrics: - # enabled: true - -# podMonitor: -# enabled: true - -# prometheusRule: -# enabled: true -# defaultRules: -# enabled: true - -# Annotations to be added to pod -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-supply-contour-ext - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-farmiso-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert \ No newline at end of file diff --git a/helm-overrides/k8s-farmiso-prd-ase1/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index eaa9f17..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,134 +0,0 @@ -fullnameOverride: opentelemetry-farmiso-prd - -mode: daemonset - -priorityClassName: "system-node-critical" - -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-supply-contour-ext - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-farmiso-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external - -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 100m - memory: 124Mi - limits: - cpu: 200m - memory: 256Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "farmiso" - team: "sre" - service: "opentelemetry-farmiso-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/k8s-farmiso-prd-ase1/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index fa154b1..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,491 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-farmiso-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "node-exporter-farmiso-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-farmiso-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index 9d85d9f..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-farmiso-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-farmiso-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "stackdriver-exporter-farmiso-prd" - env: "prd" - priority: "p0" - type: "exporter" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-farmiso-prd-0622" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'compute.googleapis.com/instance,cloudsql.googleapis.com/database,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - dedicated: vmselect-mds - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-stackdriver-exp-farmiso-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - } - - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/k8s-farmiso-prd-ase1/telegraf-operator/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/telegraf-operator/custom-values.yaml deleted file mode 100644 index a3839e1..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,151 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "farmiso" - team: "farmiso-sre" - service: "telegraf-operator-farmiso-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - dedicated: "vmselect-mds" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false \ No newline at end of file diff --git a/helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml deleted file mode 100644 index fc08e18..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml +++ /dev/null @@ -1,296 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-farmiso-prd-dr -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-farmiso-prd-dr-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent-dr - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-farm-fmsre-vmagent-prd@meesho-farmiso-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - - https://vminsert-infra-prd-dr.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "vmagent-farmiso-prd" - env: "prd" - priority: "p0" - type: "vmagent-dr" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-farmiso-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-farmiso-prd-dr.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 3 - memory: 2Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent-dr" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml deleted file mode 100644 index b940892..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml +++ /dev/null @@ -1,272 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 1 - -fullnameOverride: vmagent-farmiso-prd-fb -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-farmiso-prd-fb-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - -# Remove topology spread constraints -topologySpreadConstraints: [] - -# Modify deployment strategy -deployment: - enabled: true - -# Remove HPA -horizontalPodAutoscaler: - enabled: false - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-infr-sre-vmagent-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -# Remove pod disruption budget -podDisruptionBudget: - enabled: false - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-farmiso.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vmsinglenode-infra-prd.victoriametrics.svc.clusterset.local:8428/api/v1/write - - https://vmsinglenode-infra-prd.meeshogcp.in/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "farmiso-fb" - team: "sre" - service: "vmagent-farmiso-prd-fb" - env: "prd" - priority: "p0" - type: "vmagent-fb" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-farmiso-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-farmiso-prd-fb.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -# Modify resources -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - requests: - cpu: 2 - memory: 1Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml deleted file mode 100644 index c3a5d54..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert-dr - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert-dr - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-farmiso-prd-dr - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/farmiso/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-farmiso-prd-dr.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-farmiso-prd-dr" - env: "prd" - priority: "p0" - type: "vmalert-dr" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - dedicated: "vmagent-dr" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml deleted file mode 100644 index 62ea7f6..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-farmiso-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-farmiso-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-farmiso-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-farmiso-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/farmiso/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-farmiso-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "farmiso" - team: "sre" - service: "vmalert-farmiso-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-alert/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-alert/custom-values.yaml deleted file mode 100644 index 8ae52ec..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/victoria-metrics-alert/custom-values.yaml +++ /dev/null @@ -1,305 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-farmiso-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-farmiso-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-farmiso-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-farmiso-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config1/farmiso/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-farmiso-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-farmiso-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-farmiso-prd-ase1/victoriametrics-agent/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index a4c2b0b..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,479 +0,0 @@ -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "farmiso-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -# -- Replica count -replicaCount: 2 - -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - -# -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables -deployment: - spec: - minReadySeconds: 1200 - progressDeadlineSeconds: 1800 - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - -image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - # -- Image tag, set to `Chart.AppVersion` by default - tag: v1.133.0 # rewrites Chart.AppVersion - # -- Variant of the image to use. - # e.g. enterprise, scratch - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - -# -- Image pull secrets -imagePullSecrets: [] - -# -- Add additional DNS entries to pods hosts file. Check [official documentation](https://kubernetes.io/docs/tasks/network/customize-hosts-file-for-pods/) -hostAliases: [] -# - ip: 192.168.1.1 -# hostNames: -# - test.example.com -# - another.example.net - -# -- Override chart name -nameOverride: "" - -# -- Override resources fullname -fullnameOverride: "vm-agent-farmiso-prd" - -# -- Container working directory -containerWorkingDir: "/" - -rbac: - # -- Enables Role/RoleBinding creation - create: true - - # -- Role/RoleBinding annotations - annotations: {} - - # -- Role/RoleBinding labels - extraLabels: {} - - # -- If true and `rbac.enabled`, will deploy a Role/RoleBinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - - # -- additional rules for a role - extraRules: [] - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - # -- Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-farm-sre-vmagnt-prd-mds@meesho-farmiso-prd-0622.iam.gserviceaccount.com - } - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # -- mount API token to pod directly - automountToken: true - -# -- See `kubectl explain poddisruptionbudget.spec` for more or check [official documentation](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# -- Generates `remoteWrite.*` flags and config maps with value content for values, that are of type list of map. -# Each item should contain `url` param to pass validation. -remoteWrite: - - url: http://vm-insert-farmiso-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write - disableOnDiskQueue: false - dropSamplesOnOverload: false - -# -- VMAgent extra command line arguments -extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8429 - promscrape.config.strictParse: false # default is true - promscrape.maxScrapeSize: 1000000000 # default is 167772160 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - # promscrape.suppressScrapeErrors: true # default is false - - # Uncomment and specify the port if you want to support any of the protocols: - # https://docs.victoriametrics.com/victoriametrics/vmagent/#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for more details. -env: - - name: GOGC - value: "200" - - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth-secret - # key: password - -# -- Specify alternative source for env variables -envFrom: - [] - #- configMapRef: - # name: special-config - -# -- Extra labels for Deployment and Statefulset -extraLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "vm-agent-farmiso-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra labels for Pods only -podLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "vm-agent-farmiso-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra selector labels common for pod and service -selectorLabels: {} - -# -- Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# -- Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# -- Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -# -- Extra containers to run in a pod with vmagent -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -# -- Init containers for vmagent -initContainers: - [] - # - name: example - # image: example-image - -# -- Security context to be added to pod -podSecurityContext: - enabled: true - # fsGroup: 2000 - -# -- Security context to be added to pod's containers -securityContext: - enabled: true - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - # -- Enable agent service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - extraLabels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) for details - externalIPs: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8429 - # -- Target port - targetPort: http - # nodePort: 30000 - # -- Service type - type: ClusterIP - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - -ingress: - # -- Enable deployment of ingress for agent - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-agent-farmiso-prd.meeshogcp.in - path: - - / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - - # -- Ingress controller class name - ingressClassName: "nginx-internal" - - # -- Ingress path type - pathType: Prefix - -# -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) -resources: - requests: - cpu: 4 - memory: 8Gi - -# -- Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) -nodeSelector: - dedicated: "vmagent-mds" - -# -- Node tolerations for server scheduling to nodes with taints. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-mds" - effect: "NoSchedule" - -# -- Pod topologySpreadConstraints -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# -- Pod affinity -affinity: {} - -# -- VMAgent [scraping configuration](https://docs.victoriametrics.com/victoriametrics/vmagent/#how-to-collect-metrics-in-prometheus-format) -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-farmiso-prd" - -# -- Priority class to be assigned to the pod(s) -priorityClassName: "" - -# -- Enable the host network -hostNetwork: false - -serviceMonitor: - # -- Enable deployment of Service Monitor for server component. This is Prometheus operator object - enabled: false - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Service Monitor relabelings - relabelings: [] - # -- Basic auth params for Service Monitor - basicAuth: {} - # -- Service Monitor metricRelabelings - metricRelabelings: [] - # -- Service Monitor targetPort - targetPort: http - # interval: 15s - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - -# -- Empty dir configuration for a case, when persistence is disabled -emptyDir: {} - -persistentVolume: - # -- Create/use Persistent Volume Claim for server component. Empty dir if false - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Size of the volume. Should be calculated based on the logs you send and retention policy you set. - size: 10Gi - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume additional labels - extraLabels: {} - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Horizontal Pod Autoscaling. -# Note that it is not intended to be used for vmagents which perform scraping. -# In order to scale scraping vmagents check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) -horizontalPodAutoscaling: - # -- Use HPA for vmagent - enabled: false - # -- Maximum replicas for HPA to use to to scale vmagent - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale vmagent - minReplicas: 1 - # -- Metric for HPA to use to scale vmagent - metrics: [] - -# -- VMAgent scrape configuration -config: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -probe: - # -- Readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - # -- Liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - # -- Startup probe - startup: {} - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -allowedMetricsEndpoints: - - /metrics - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-farmiso-prd-ase1/victoriametrics-insert/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/victoriametrics-insert/custom-values.yaml deleted file mode 100644 index 9df5f0f..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/victoriametrics-insert/custom-values.yaml +++ /dev/null @@ -1,411 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: false - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- IDs of vmstorage nodes to exclude from writing - excludeStorageIDs: [] - # -- Override default `app` label name - name: vminsert - # -- VMInsert strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - # override Chart.AppVersion - tag: v1.107.0-cluster - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMInsert http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-farmiso-prd - # -- Extra command line arguments for vminsert component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8480 - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-farmiso-prd-0.vm-storage-farmiso-prd.victoriametrics.svc:8400" - - "vm-storage-farmiso-prd-1.vm-storage-farmiso-prd.victoriametrics.svc:8400" - - "vm-storage-farmiso-prd-2.vm-storage-farmiso-prd.victoriametrics.svc:8400" - - "vm-storage-farmiso-prd-3.vm-storage-farmiso-prd.victoriametrics.svc:8400" - - "vm-storage-farmiso-prd-4.vm-storage-farmiso-prd.victoriametrics.svc:8400" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "vm-insert-farmiso-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Pod's additional labels - podLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "vm-insert-farmiso-prd" - env: "prd" - priority: "p0" - type: "vminsert" - terminationGracePeriodSeconds: 30 - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - - # -- Readiness & Liveness probes - probe: - # -- VMInsert readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert startup probe - startup: {} - - # -- Relabel configuration - relabel: - enabled: false - config: [] - # -- Use existing configmap if specified - # otherwise .config values will be used. Relabel config **should** reside under `relabel.yml` key - configMap: "" - - # Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 2 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vminsert - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vminsert - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - dedicated: "vminsert-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 2 - memory: 4Gi - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - service: - # -- Create VMInsert service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here]( https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) for details - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-insert-farmiso-prd.meeshogcp.in - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - - # -- Ingress controller class name - ingressClassName: nginx-internal - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for insert component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/insert" .Values.vminsert }}' - - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vminsert component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-farmiso-prd-ase1/victoriametrics-select/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/victoriametrics-select/custom-values.yaml deleted file mode 100644 index 841dba2..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/victoriametrics-select/custom-values.yaml +++ /dev/null @@ -1,448 +0,0 @@ - -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Enable split services for vmselect (extra Service objects will be created by templates/service-split.yaml) - splitService: true - externalService: - enabled: true - name: vmselect-farmiso-prd-proxy - # -- Override default `app` label name - clusternativeService: - enabled: false - name: "" - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # override Chart.AppVersion - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMSelect http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-farmiso-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - # -- Extra command line arguments for vmselect component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8481 - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 180s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 96 - storageNode: - - "vm-storage-farmiso-prd-0.vm-storage-farmiso-prd.victoriametrics.svc:8401" - - "vm-storage-farmiso-prd-1.vm-storage-farmiso-prd.victoriametrics.svc:8401" - - "vm-storage-farmiso-prd-2.vm-storage-farmiso-prd.victoriametrics.svc:8401" - - "vm-storage-farmiso-prd-3.vm-storage-farmiso-prd.victoriametrics.svc:8401" - - "vm-storage-farmiso-prd-4.vm-storage-farmiso-prd.victoriametrics.svc:8401" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "vm-select-farmiso-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # -- Pod's additional labels - podLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "vm-select-farmiso-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Readiness & Liveness probes - probe: - # -- VMSelect readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 15 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 3 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vmselect - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vmselect - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - dedicated: "vmselect-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 3 - memory: 8Gi - - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Create VMSelect service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: false - - # -- Ingress annotations - annotations: {} - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-select-farmiso.prd.meesho.int - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # -- Ingress controller class name - ingressClassName: contour-internal-0 - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for select component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/select" .Values.vmselect }}' - - # -- vmselect mode: deployment, daemonSet - mode: deployment - - # -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables - deployment: - spec: - # -- VMSelect strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - - # -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables - statefulSet: - enabled: false - spec: - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume extra labels - extraLabels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmselect component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: false - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - diff --git a/helm-overrides/k8s-farmiso-prd-ase1/victoriametrics-storage/custom-values.yaml b/helm-overrides/k8s-farmiso-prd-ase1/victoriametrics-storage/custom-values.yaml deleted file mode 100644 index 898d8a9..0000000 --- a/helm-overrides/k8s-farmiso-prd-ase1/victoriametrics-storage/custom-values.yaml +++ /dev/null @@ -1,363 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - enabled: false - -vminsert: - enabled: false - -vmauth: - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- Override default `app` label name - name: vmstorage - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Variant of the image to use. e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMStorage http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-farmiso-prd - - # -- Additional environment variables (ex.: secret tokens, flags). Check https://docs.victoriametrics.com/victoriametrics/#environment-variables for details - env: [] - # -- Specify alternative source for env variables - envFrom: [] - # -- Data retention period. Possible units character: h(ours), d(ays), w(eeks), y(ears), if no unit character specified - month. The minimum retention period is 24h. - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - httpListenAddr: :8482 - loggerTimezone: "Asia/Kolkata" - dedup.minScrapeInterval: 60s - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage-n4d" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage-n4d" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - # -- Use an alternate scheduler, e.g. "stork". Check https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ for details - # schedulerName: - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Override Persistent Volume Claim name - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are https://kubernetes.io/docs/concepts/storage/persistent-volumes/ - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume extra labels - extraLabels: {} - # -- Storage class name. Will be empty if not set - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume - size: 250Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "vm-storage-farmiso-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - # -- Pod's additional labels - podLabels: - bu: "farmiso" - team: "farmiso-sre" - service: "vm-storage-farmiso-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Details are https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 6 - memory: 51Gi - - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: - enabled: false - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - podSecurityContext: - enabled: false - service: - enabled: true - # -- Service annotations - annotations: {} - # -- Service ClusterIP - clusterIP: None - # -- Service type - type: ClusterIP - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmselect - vmselectPort: 8401 - # -- Extra service ports - extraPorts: [] - # -- Health check node port for a service. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilyPolicy: "" - # -- List of service IP families. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilies: [] - # -- Traffic Distribution. Check https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution - trafficDistribution: "" - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - minReadySeconds: 5 - # -- Readiness probes - probe: - # -- VMStorage readiness probe - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMStorage startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmstorage component - enabled: false - # -- Maximum replicas for HPA to use to to scale the vmstorage component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmstorage component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmstorage component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: - scaleDown: - selectPolicy: Disabled - - vmbackupmanager: - # -- Enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" \ No newline at end of file diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/README.md b/helm-overrides/k8s-ml-platform-prd-ase1/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/cert-manager/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/cert-manager/custom-values.yaml deleted file mode 100644 index 2cc7620..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/cert-manager/custom-values.yaml +++ /dev/null @@ -1,129 +0,0 @@ -global: - logLevel: 2 - rbac: - create: true - priorityClassName: "high-priority" -installCRDs: false - -crds: - enabled: true - keep: true - -# Cert-manager Controller -replicaCount: 1 -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-controller - tag: v1.20.1 - pullPolicy: IfNotPresent - -nodeSelector: - dedicated: ml-platform-devops - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "ml-platform-devops" - effect: "NoSchedule" - -resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 500m - memory: 512Mi - -# Webhook Configuration -webhook: - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-webhook - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: ml-platform-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "ml-platform-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# CA Injector Configuration -cainjector: - enabled: true - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-cainjector - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: ml-platform-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "ml-platform-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# ACME Solver Configuration -acmesolver: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-acmesolver - tag: v1.20.1 - pullPolicy: IfNotPresent - -# Startup API Check -startupapicheck: - enabled: true - timeout: 1m - backoffLimit: 4 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-startupapicheck - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: ml-platform-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "ml-platform-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - -# Prometheus Monitoring -prometheus: - enabled: true - servicemonitor: - enabled: false - interval: 60s - scrapeTimeout: 30s - labels: - prometheus: cert-manager diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/conntrack-adjuster/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 7b64b15..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - # Maximum number of conntrack entries - max: 2097152 - # Hash size for conntrack - hashsize: 524288 - # Sleep interval between adjustments (seconds) - sleepInterval: 30 - -# Enable additional matchExpressions -# addExtraMatchExpressions: true - -# Additional matchExpressions to append -# additionalMatchExpressions: -# - key: node-role -# operator: In -# values: -# - "worker" -# - "ingress" -# - key: environment -# operator: In -# values: -# - "production" -# - "staging" \ No newline at end of file diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/contour-cert-checker/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/contour-cert-checker/custom-values.yaml deleted file mode 100644 index d181ac0..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=k8s-ml-platform-prd-ase1"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - dedicated: ml-platform-devops - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: ml-platform-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/contour-internal-0/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/contour-internal-0/custom-values.yaml deleted file mode 100644 index f3f8d58..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,94 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 60s - max-connection-duration: 600s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: ml-platform - team: ml-platform-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 500m - memory: 3Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: ml-platform - team: ml-platform-devops - env: prd - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 5% - maxUnavailable: 0 - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 8Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-mlp-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/contour-internal-1/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/contour-internal-1/custom-values.yaml deleted file mode 100644 index 6e3495d..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,110 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 65s - connection-shutdown-grace-period: 400s - max-connection-duration: 540s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: ml-platform - team: ml-platform-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 4 - memory: 8Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: ml-platform - team: ml-platform-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - logLevel: error - terminationGracePeriodSeconds: 500 - autoscaling: - behaviour: - scaledown: - policies: - - periodseconds: 500 - type: Pods - value: 1 - selectpolicy: Min - stabilizationWindowSeconds: 600 - scaleup: - policies: - - periodseconds: 15 - type: Pods - value: 2 - selectpolicy: Max - stabilizationWindowSeconds: 60 - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "35" - targetMemory: "40" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 7 - memory: 8Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int1-mlp-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/coredns/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/coredns/custom-values.yaml deleted file mode 100644 index e1b00d3..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/coredns/custom-values.yaml +++ /dev/null @@ -1,27 +0,0 @@ -replicaCount: 16 - -labels: - bu: ml-platform - team: ml-platform-devops - env: prd - -clusterIP: 10.137.18.2 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "ml-platform-devops" - effect: "NoSchedule" - -nodeSelector: - dedicated: ml-platform-devops - kubernetes.io/os: linux \ No newline at end of file diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/external-secrets/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/external-secrets/custom-values.yaml deleted file mode 100644 index ad18a89..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/external-secrets/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Additional service annotations - annotations: {} - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: ml-platform-devops - nodeSelector: - dedicated: ml-platform-devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: ml-platform-devops - nodeSelector: - dedicated: ml-platform-devops - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: ml-platform-devops - nodeSelector: - dedicated: ml-platform-devops diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/flagger/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/flagger/custom-values.yaml deleted file mode 100644 index ec1a9e0..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/flagger/custom-values.yaml +++ /dev/null @@ -1,62 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - dedicated: ml-platform-devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: ml-platform-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: ml-platform-devops - bu: infra - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-ml-platform-rollout-service.prd-ml-platform-rollout-service.svc.cluster.local" - -rolloutService: - enabled: true - apps: diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/fluentd/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/fluentd/custom-values.yaml deleted file mode 100644 index a5a24d6..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/fluentd/custom-values.yaml +++ /dev/null @@ -1,717 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-mlp-mlpsre-fluentd-prd@meesho-ml-platform-prd-0625.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: ml-platform - team: sre - type: fluentd - service: fluentd-ml-platform-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: ml-platform - team: sre - type: fluentd - service: fluentd-ml-platform-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: [] -# - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/keda/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/keda/custom-values.yaml deleted file mode 100644 index 119ec61..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/keda/custom-values.yaml +++ /dev/null @@ -1,46 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: ml-platform-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "ml-platform-devops" - effect: "NoSchedule" - podLabels: - bu: "ml-platform" - team: "ml-platform-devops" - metricsAdapter: - bu: "ml-platform" - team: "ml-platform-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 2000Mi - requests: - cpu: 300m - memory: 450Mi - # -- Manage [resource request & limits] of KEDA metrics apiserver pod - metricServer: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 100m - memory: 200Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 100m - memory: 180Mi - requests: - cpu: 60m - memory: 120Mi - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' \ No newline at end of file diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/kube-dns/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/kube-dns/custom-values.yaml deleted file mode 100644 index 0adf8b9..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.137.18.2"],"prd.mrouter.int.svc.cluster.local":["10.137.18.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.137.18.2"]} diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/kube-events/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/kube-events/custom-values.yaml deleted file mode 100644 index d600c7d..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-ml-platform-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - dedicated: ml-platform-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: ml-platform-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: ml-platform-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: ml-platform-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: ml-platform-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: ml-platform-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 0e4d1f6..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-ml-platform-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-ml-platform-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "ml-platform" - team: "ml-platform-sre" - service: "kube-state-metrics-ml-platform-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "vmselect" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 17m - memory: 1.5Gi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index e16250d..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,133 +0,0 @@ -fullnameOverride: opentelemetry-ml-platform-prd - -mode: daemonset - -# priorityClassName: "system-node-critical" -# Removing since dataengg pods are going into pending state -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-ml-platform-contour-ext - operator: NotIn - values: - - dedicated - - key: p-ml-platform-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-ml-platform-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-ml-platform-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-ml-platform-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external - -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 100m - memory: 124Mi - limits: - cpu: 200m - memory: 256Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "ml-platform" - team: "sre" - service: "opentelemetry-ml-platform-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index a0466ea..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,496 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-dataengg-prd@meesho-dataengg-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "ml-platform" - team: "ml-platform-sre" - service: "node-exporter-ml-platform-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index 85723a9..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-ml-platform-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-ml-platform-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "ml-platform" - team: "ml-platform-sre" - service: "stackdriver-exporter-ml-platform-prd" - env: "prd" - priority: "p0" - type: "exporter" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-ml-platform-prd-0625" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,logging.googleapis.com/user,run.googleapis.com/,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - dedicated: vmselect - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-mlp-mlpsre-stackdriver-prd@meesho-ml-platform-prd-0625.iam.gserviceaccount.com - } - - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/telegraf-operator/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/telegraf-operator/custom-values.yaml deleted file mode 100644 index 5df6e04..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,150 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "ml-platform" - team: "ml-platform-sre" - service: "telegraf-operator-ml-platform-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - dedicated: "vmselect" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-agent/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index 15dba30..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,295 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-ml-platform-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-ml-platform-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - minReadySeconds: 30 - progressDeadlineSeconds: 60 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-mlp-mlpsre-vmagent-prd@meesho-ml-platform-prd-0625.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-ml-platform.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-ml-platform-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "ml-platform" - team: "ml-platform-sre" - service: "vmagent-ml-platform-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-ml-platform-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: contour-internal-1 - annotations: {} - # nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - # nginx.ingress.kubernetes.io/rewrite-target: / - # nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-ml-platform-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 6 - memory: 4Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml deleted file mode 100644 index 1bb3ca5..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-ml-platform-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-ml-platform-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-ml-platform-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-ml-platform-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/datascience/ml-platform/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-ml-platform-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-ml-platform-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-alert/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-alert/custom-values.yaml deleted file mode 100644 index 247e451..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-alert/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-ml-platform-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-ml-platform-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-ml-platform-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-ml-platform-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/ml-platform/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-ml-platform-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-ml-platform-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - dedicated: "vmselect" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-insert/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index 63a1f98..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,234 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-ml-platform-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-ml-platform-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "ml-platform" - team: "ml-platform-sre" - service: "vminsert-ml-platform-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 20 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 10 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vminsert" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 3 - memory: 1.5Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: true - # -- Ingress annotations - annotations: - # nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - # nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-ml-platform-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: contour-internal-0 - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-select/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index f81a758..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,293 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-ml-platform-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-ml-platform-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 96 - clusternative.maxConcurrentRequests: 256 - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "ml-platform" - team: "ml-platform-sre" - service: "vmselect-ml-platform-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 4 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "vmselect" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 6 - memory: 12Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-ml-platform-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-storage/custom-values.yaml b/helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index 0949c96..0000000 --- a/helm-overrides/k8s-ml-platform-prd-ase1/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,316 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-ml-platform-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: standard-rwo - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 250Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "ml-platform" - team: "ml-platform-sre" - service: "vmstorage-ml-platform-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 20 - memory: 193Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.93.7-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-ml-platform-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-sec-admin-ase1/coredns/custom-values.yaml b/helm-overrides/k8s-sec-admin-ase1/coredns/custom-values.yaml deleted file mode 100644 index 183682c..0000000 --- a/helm-overrides/k8s-sec-admin-ase1/coredns/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -replicaCount: 1 - -labels: - bu: infra - team: sec - env: admin - -clusterIP: 10.137.28.2 - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" - -nodeSelector: - dedicated: devops - kubernetes.io/os: linux \ No newline at end of file diff --git a/helm-overrides/k8s-sec-admin-ase1/deepfence-console/custom-values.yaml b/helm-overrides/k8s-sec-admin-ase1/deepfence-console/custom-values.yaml deleted file mode 100644 index f5baf9e..0000000 --- a/helm-overrides/k8s-sec-admin-ase1/deepfence-console/custom-values.yaml +++ /dev/null @@ -1,534 +0,0 @@ -# Default values for deepfence-console. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -nameOverride: "" -fullnameOverride: "sec-admin-deepfence-console" - -global: - imageRepoPrefix: "quay.io" - # imageRepoPrefix: "docker.io" - # this image tag is used everywhere for console services - # to override set tag at service level - imageTag: 2.1.0 - storageClass: "standard-rwo" - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" - -imagePullSecret: - # Specifies whether a image pull secret should be created - create: true - registry: "quay.io" - # registry: "https://index.docker.io/v1/" - username: "deepfenceio+meesho_com" - password: "KM8X1CZGMXAJ4IRS8DDM8RIPZDFAYGBUGI78YFEJXBA1RCNY0GIBW10N7KNHOYS2" - # The name of the imagePullSecret to use. - # If not set and create is true, a name is generated using the fullname template - name: "" - -kafka: - # Specifies whether a kafka cluster should be created - create: true - # if create false provide name of the existing secret - # secret format refer templates/console-secrets/kafka.yaml - secretName: "" - # if create true then below values are used to create kafka cluster - replicaCount: 1 # recommended 3 for high availability kafka - image: - repository: deepfenceio/deepfence_kafka_broker - pullPolicy: Always - # Overrides the image tag whose default is .global.imageTag - # tag: 2.0.1 - config: - # required, recommended to generate new UUID using kafka-storage tool - STORAGE_UUID: hNQ55qppT5GGybF52ZGlOQ - storageClass: "" - volumeSize: 50G - resources: - limits: - cpu: 4000m - memory: 8192Mi - requests: - cpu: 500m - memory: 1024Mi - podAnnotations: {} - podSecurityContext: {} - securityContext: {} - nodeSelector: {} - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepfence - affinity: {} - -postgres: - # Specifies whether a postgres database instance should be created - create: true - # if create false provide name of the existing secret - # secret format refer templates/deepfence-console-secrets/postgres.yaml - secretName: "" - # if create true then below values are used to create postgres database instance - secrets: - POSTGRES_PASSWORD: deepfence - POSTGRES_USER: deepfence - POSTGRES_DB: users - image: - repository: deepfenceio/deepfence_postgres - pullPolicy: Always - # Overrides the image tag whose default is .global.imageTag - # tag: 2.0.1 - storageClass: "" - volumeSize: 50G - resources: - limits: - cpu: 2000m - memory: 2048Mi - requests: - cpu: 200m - memory: 512Mi - podAnnotations: {} - podSecurityContext: {} - securityContext: {} - nodeSelector: {} - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepfence - affinity: {} - -redis: - # Specifies whether a postgres database instance should be created - create: true - # if create false provide name of the existing secret - # secret format refer templates/console-secrets/redis.yaml - secretName: "" - # if create true then below values are used to create postgres database instance - image: - repository: deepfenceio/deepfence_redis - pullPolicy: Always - # Overrides the image tag whose default is .global.imageTag - # tag: 2.0.1 - storageClass: "" - volumeSize: 10G - resources: - limits: - cpu: 1000m - memory: 2048Mi - requests: - cpu: 100m - memory: 128Mi - podAnnotations: {} - podSecurityContext: {} - securityContext: {} - nodeSelector: {} - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepfence - affinity: {} - -fileserver: - # Specifies whether a file server instance should be created - # set this to false if using S3 - create: true - # if create false provide name of the existing secret - # secret format refer templates/console-secrets/minio.yaml - secretName: "" - # if create true then below values are used to create postgres database instance - secrets: - MINIO_ROOT_USER: deepfence - MINIO_ROOT_PASSWORD: deepfence - image: - repository: deepfenceio/deepfence_file_server - pullPolicy: Always - # Overrides the image tag whose default is .global.imageTag - # tag: 2.0.1 - storageClass: "" - volumeSize: 50G - resources: - limits: - cpu: 2000m - memory: 4096Mi - requests: - cpu: 100m - memory: 128Mi - podAnnotations: {} - podSecurityContext: {} - securityContext: {} - nodeSelector: {} - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepfence - affinity: {} - -# these values are used if fileserver.create=false -aws_s3_buckets: - # Specifies whether secret should be created - create: false - # if create false provide name of the existing secret - # secret format refer templates/deepfence-console-secrets/s3.yaml - secretName: "" - # public bucket with read permisons on objects for hosting vulnerability database - # S3 bucket permissions {"Version":"2012-10-17","Statement":[{"Sid":"database","Effect":"Allow","Principal":"*","Action":"s3:GetObject","Resource":["arn:aws:s3:::/database/*","arn:aws:s3:::/database"]}]} - vulnerability_db_bucket: "" - # prvate bucket to host reports, sbom, etc. - data_bucket: "" - # aws credentials to access buckets - access_key_id : "" - secret_access_key: "" - # region where the buckets are hosted ex: ap-south-1 - region: "" - -neo4j: - # Specifies whether a neo4j database instance should be created - create: true - # if create false provide name of the existing secret - # secret format refer templates/console-secrets/neo4j.yaml - secretName: "" - # if create true then below values are used to create neo4j database instance - secrets: - # format should be username/password - NEO4J_AUTH: neo4j/e16908ffa5b9f8e9d4ed - # To enable periodic backup of neo4j database to S3, please set the values below - # AWS_ACCESS_KEY: "" - # AWS_SECRET_KEY: "" - # DF_REMOTE_BACKUP_ROOT: "" # S3 bucket name - config: - NEO4J_dbms_memory_pagecache_size: 2600m - NEO4JLABS_PLUGINS: '["apoc"]' - image: - repository: deepfenceio/deepfence_neo4j - pullPolicy: Always - # tag: 2.0.1 - storageClass: "" - volumeSize: 50G - resources: - limits: - cpu: 4000m - memory: 16Gi - requests: - cpu: 1000m - memory: 2048Mi - podAnnotations: {} - podSecurityContext: {} - securityContext: {} - nodeSelector: {} - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepfence - affinity: {} - -# ingress for console -ingress: - enable: false - ## name of the ingress class for ingress provider installed on the cluster, cannot be empty - ## Example: nginx - class: nginx - ## host example: threat.example.com - host: "" - ## annotations to customize ingress - annotations: - ## nginx ingress annotations - ## https://kubernetes.github.io/ingress-nginx/user-guide/nginx-configuration/ - nginx.ingress.kubernetes.io/backend-protocol: HTTPS - nginx.ingress.kubernetes.io/force-ssl-redirect: "true" - nginx.ingress.kubernetes.io/proxy-body-size: 200m - - ## aws alb annotations - ## aws load balancer controller needs to be installed on the cluster for these annotations to work - ## documentation aws load balancer controller https://kubernetes-sigs.github.io/aws-load-balancer-controller/v2.4/guide/ingress/annotations/ - # alb.ingress.kubernetes.io/actions.ssl-redirect: '{"Type": "redirect", "RedirectConfig": { "Protocol": "HTTPS", "Port": "443", "StatusCode": "HTTP_301"}}' - # alb.ingress.kubernetes.io/backend-protocol: HTTPS - ## arn of the certificate available on aws certificate manager - # alb.ingress.kubernetes.io/certificate-arn: "" - # alb.ingress.kubernetes.io/listen-ports: '[{"HTTPS":443}, {"HTTP":80}]' - # alb.ingress.kubernetes.io/scheme: internet-facing - # alb.ingress.kubernetes.io/target-group-attributes: stickiness.enabled=true,stickiness.lb_cookie.duration_seconds=3600 - # alb.ingress.kubernetes.io/target-type: ip - -router: - replicaCount: 1 - image: - repository: deepfenceio/deepfence_router - pullPolicy: Always - # Overrides the image tag whose default is .global.imageTag - # tag: 2.0.1 - forceHttpsRedirect: true - podAnnotations: {} - podSecurityContext: {} - securityContext: {} - service: - ## useful if deepfence-router chart is not installed - create: false - # useful for configuring loadbalancer options on supported clouds - annotations: {} - ## service.beta.kubernetes.io/do-loadbalancer-enable-proxy-protocol: "true" - type: ClusterIP # set service type to cluster ip and enable ingress if available - httpsPort: 443 - httpPort: 80 - resources: - limits: - cpu: 3000m - memory: 4096Mi - requests: - cpu: 100m - memory: 128Mi - autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 3 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - nodeSelector: {} - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepfence - affinity: {} - # Use custom ssl certificate for Deepfence UI - # custom certificates can be configured using two options - # existing secret or base64 encoded cert and key string - # provide one off the two options to configure custom certificates - tls: - # provide secret name which contains tls cert and key - # reference: https://kubernetes.io/docs/concepts/configuration/secret/#tls-secrets - # make sure to create secret in the same namespace as that of the console - secretName: "" - # embed given cert and key as secret and mount to router pod - # provide certificate and key in below example format - # cert: |- - # -----BEGIN CERTIFICATE----- - # MIIFCTCCAvGgAwIBAgIUNshy8GFTjfUR7inZ1JCcN+tDuh4wDQYJKoZIhvcNAQEL - # ..... - # BMepE4d9+TQFcPQ/OKSlP8FB2nPKZJdM+JlXDFWqeKvbdYS4QErRLd33qUmq - # -----END CERTIFICATE----- - # key: |- - # -----BEGIN PRIVATE KEY----- - # MIIJQQIBADANBgkqhkiG9w0BAQEFAASCCSswggknAgEAAoICAQDECeUraonCz/89 - # ..... - # bHEvWp7ugCTFhurM+lla0d+ElDO2 - # -----END PRIVATE KEY----- - cert: "" - key: "" - -server: - replicaCount: 1 - image: - repository: deepfenceio/deepfence_server - pullPolicy: Always - # Overrides the image tag whose default is .global.imageTag - # tag: 2.0.1 - podAnnotations: {} - podSecurityContext: {} - securityContext: {} - service: - type: ClusterIP - port: 8080 - internalPort: 8081 - resources: - limits: - cpu: 3000m - memory: 4096Mi - requests: - cpu: 250m - memory: 256Mi - autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 3 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - nodeSelector: {} - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepfence - affinity: {} - -worker: - replicaCount: 1 - image: - repository: deepfenceio/deepfence_worker - pullPolicy: Always - # Overrides the image tag whose default is .global.imageTag - # tag: 2.0.1 - podAnnotations: {} - podSecurityContext: {} - securityContext: {} - service: - type: ClusterIP - port: 8080 - resources: - limits: - cpu: 2000m - memory: 8000Mi - requests: - cpu: 250m - memory: 256Mi - autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 3 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - nodeSelector: {} - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepfence - affinity: {} - -ingester: - replicaCount: 1 - image: - repository: deepfenceio/deepfence_worker - pullPolicy: Always - # Overrides the image tag whose default is .global.imageTag - # tag: 2.0.1 - podAnnotations: {} - podSecurityContext: {} - securityContext: {} - service: - type: ClusterIP - port: 8080 - resources: - limits: - cpu: 2000m - memory: 4096Mi - requests: - cpu: 100m - memory: 128Mi - autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 3 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - nodeSelector: {} - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepfence - affinity: {} - -scheduler: - image: - repository: deepfenceio/deepfence_worker - pullPolicy: Always - # Overrides the image tag whose default is .global.imageTag - # tag: 2.0.1 - service: - type: ClusterIP - port: 8080 - resources: - limits: - cpu: 1000m - memory: 512Mi - requests: - cpu: 100m - memory: 128Mi - podAnnotations: {} - podSecurityContext: {} - securityContext: {} - nodeSelector: {} - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepfence - affinity: {} - -ui: - replicaCount: 1 - image: - repository: deepfenceio/deepfence_ui - pullPolicy: Always - # Overrides the image tag whose default is .global.imageTag - # tag: 2.0.1 - service: - type: ClusterIP - port: 8081 - resources: - limits: - cpu: 1000m - memory: 512Mi - requests: - cpu: 100m - memory: 128Mi - autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 3 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - podAnnotations: {} - podSecurityContext: {} - securityContext: {} - nodeSelector: {} - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepfence - affinity: {} - -console_agents: - enabled: false - cluster_name: "tm-cluster" - enableGraphReport: true - userDefinedTags: "" - instanceIdSuffix: "N" - mountContainerRuntimeSocket: - dockerSock: false - # Change if socket path is not the following - dockerSockPath: "/var/run/docker.sock" - containerdSock: true - # Change if socket path is not the following - containerdSockPath: "/run/containerd/containerd.sock" - crioSock: false - # Change if socket path is not the following - crioSockPath: "/var/run/crio/crio.sock" - podmanSock: false - # Change if socket path is not the following - podmanSockPath: "/run/podman/podman.sock" - agent: - image: - repository: deepfenceio/deepfence_agent - pullPolicy: Always - # Overrides the image tag whose default is .global.imageTag - # tag: 2.0.1 - resources: - requests: - cpu: 150m - memory: 512Mi - limits: - cpu: 1500m - memory: 2048Mi - podAnnotations: {} - podSecurityContext: {} - securityContext: {} - nodeSelector: {"kubernetes.io/os": "linux"} - tolerations: - - operator: "Exists" - effect: "NoSchedule" - - operator: "Exists" - effect: "NoExecute" - affinity: {} diff --git a/helm-overrides/k8s-sec-admin-ase1/deepfence-router/custom-values.yaml b/helm-overrides/k8s-sec-admin-ase1/deepfence-router/custom-values.yaml deleted file mode 100644 index 52c50b1..0000000 --- a/helm-overrides/k8s-sec-admin-ase1/deepfence-router/custom-values.yaml +++ /dev/null @@ -1,175 +0,0 @@ -# Default values for deepfence-router. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -nameOverride: "" -fullnameOverride: "sec-admin-deepfence-router" - -# Configure port for browser / agents -managementConsolePort: "443" - -service: - name: sec-admin-deepfence-console-router - # Select the type of service to be used. - # When exposing the service in an on premisses Kubernetes cluster, select NodePort as type - # Also, possible to use Ingress as type when ingress controller is installed - type: Ingress # LoadBalancer/NodePort/Ingress/ClusterIP - # NodePort configuration. Only used when the service type is NodePort - nodePortHttps: "" - nodePortHttp: "" - # Using static ip address for load balancer - # - Google Cloud: https://cloud.google.com/kubernetes-engine/docs/tutorials/configuring-domain-name-static-ip - # loadBalancerIP: "1.2.3.4" - # - Azure: https://docs.microsoft.com/en-us/azure/aks/static-ip - # loadBalancerIP: "1.2.3.4" - loadBalancerIP: "" - # If loadBalancerType is "external", we recommend setting loadBalancerSourceRanges - # to the ip address / CIDR ranges of your laptop's ip or corporate CIDR range. - # If this is set empty, ports 443 and 80 will be open to the public internet. - # Example: ["143.231.0.0/16","210.57.79.18/32"] - #loadBalancerSourceRanges: ["0.0.0.0/0"] - - # externalIPs: When kubernetes is not cloud managed, add public ip addresses of kubernetes nodes to externalIPs - externalIPs: [] - externalTrafficPolicy: "Cluster" - - annotations: - ## aws - ## as default aws creates classic load balancer, to change to nlb use below annotation - ## https://kubernetes.io/docs/concepts/services-networking/service/#aws-nlb-support - #service.beta.kubernetes.io/aws-load-balancer-type: "nlb" - - ## Static ip for NLB - ## https://docs.aws.amazon.com/eks/latest/userguide/network-load-balancing.html - ## Example: "eipalloc-0123456789abcdefg,eipalloc-0123456789hijklmn" - # service.beta.kubernetes.io/aws-load-balancer-eip-allocations: "" - - ## ACM SSL certificate for AWS Classic LoadBalancer - ## This cannot be set if aws-load-balancer-eip-allocations is set - ## https://kubernetes.io/docs/concepts/services-networking/service/#ssl-support-on-aws - ## https://aws.amazon.com/premiumsupport/knowledge-center/terminate-https-traffic-eks-acm/ - ## Example: "arn:aws:acm:{region}:{user id}:certificate/{id}" - # service.beta.kubernetes.io/aws-load-balancer-ssl-cert: "" - # service.beta.kubernetes.io/aws-load-balancer-backend-protocol: "https" - # service.beta.kubernetes.io/aws-load-balancer-ssl-ports: "443" - - ## if internal load balancer is required - ## set this based on cloud provider - - ## aws - ## https://kubernetes.io/docs/concepts/services-networking/service/#internal-load-balancer - # service.beta.kubernetes.io/aws-load-balancer-internal: "true" - - ## azure - # service.beta.kubernetes.io/azure-load-balancer-internal: "true" - - ## gcp - #networking.gke.io/load-balancer-type: "Internal" - #cloud.google.com/load-balancer-type: "Internal" - #cloud.google.com/app-protocols: '{"https-port":"HTTPS","http-port":"HTTP"}' - - ## ibm cloud - # service.kubernetes.io/ibm-load-balancer-cloud-provider-ip-type: "private" - - ## openstack - # service.beta.kubernetes.io/openstack-internal-load-balancer: "true" - - -# User can create separate k8s service for agents if required. -# One use case for this is to deploy external load balancer for browser access for management console and internal load balancer for agent communication. -createSeparateServiceForAgents: false - -agentService: - # Configuration service accessed by agents - name: deepfence-agent-router - type: LoadBalancer # LoadBalancer/NodePort/Ingress/ClusterIP - # Using static ip address for load balancer - # - Google Cloud: https://cloud.google.com/kubernetes-engine/docs/tutorials/configuring-domain-name-static-ip - # loadBalancerIP: "1.2.3.4" - # - Azure: https://docs.microsoft.com/en-us/azure/aks/static-ip - # loadBalancerIP: "1.2.3.4" - loadBalancerIP: "" - # If loadBalancerType is "external", we recommend setting loadBalancerSourceRanges to the ip address / CIDR ranges - # of your laptop's ip or corporate CIDR range. If this is set empty, ports and 80 will be open to the public internet. - # Example: ["143.231.0.0/16","210.57.79.18/32"] - loadBalancerSourceRanges: [] - # externalIPs: When kubernetes is not cloud managed, add public ip addresses of kubernetes nodes to externalIPs - externalIPs: [] - externalTrafficPolicy: "Cluster" - - annotations: - ## aws - ## as default aws creates classic load balancer, to change to nlb use below annotation - ## https://kubernetes.io/docs/concepts/services-networking/service/#aws-nlb-support - service.beta.kubernetes.io/aws-load-balancer-type: "nlb" - - ## Static ip for NLB - ## https://docs.aws.amazon.com/eks/latest/userguide/network-load-balancing.html - ## Example: "eipalloc-0123456789abcdefg,eipalloc-0123456789hijklmn" - # service.beta.kubernetes.io/aws-load-balancer-eip-allocations: "" - - ## ACM SSL certificate for AWS Classic LoadBalancer - ## This cannot be set if aws-load-balancer-eip-allocations is set - ## https://kubernetes.io/docs/concepts/services-networking/service/#ssl-support-on-aws - ## https://aws.amazon.com/premiumsupport/knowledge-center/terminate-https-traffic-eks-acm/ - ## Example: "arn:aws:acm:{region}:{user id}:certificate/{id}" - # service.beta.kubernetes.io/aws-load-balancer-ssl-cert: "" - # service.beta.kubernetes.io/aws-load-balancer-backend-protocol: "https" - # service.beta.kubernetes.io/aws-load-balancer-ssl-ports: "" - - ## https://kubernetes.io/docs/concepts/services-networking/service/#other-elb-annotations - ## if internal load balancer is required - ## set this based on cloud provider - - ## aws - ## https://kubernetes.io/docs/concepts/services-networking/service/#internal-load-balancer - # service.beta.kubernetes.io/aws-load-balancer-internal: "true" - - ## azure - # service.beta.kubernetes.io/azure-load-balancer-internal: "true" - - ## gcp - # networking.gke.io/load-balancer-type: "Internal" - # cloud.google.com/load-balancer-type: "Internal" - # cloud.google.com/app-protocols: '{"https-port":"HTTPS","http-port":"HTTP"}' - - ## ibm cloud - # service.kubernetes.io/ibm-load-balancer-cloud-provider-ip-type: "private" - - ## openstack - # service.beta.kubernetes.io/openstack-internal-load-balancer: "true" - - -# ingress configuration for console -ingress: - ## name of the ingress class for ingress provider installed on the cluster, cannot be empty - ## Example: nginx - class: nginx-internal - ## host example: threat.example.com - host: "deepfence-sec-admin.meeshogcp.in" - ## annotations to customize ingress - annotations: - ## nginx ingress annotations - nginx.ingress.kubernetes.io/ssl-redirect: "false" - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - #nginx.ingress.kubernetes.io/rewrite-target: / - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-sec-admin"}}}' - nginx.ingress.kubernetes.io/backend-protocol: "HTTPS" - - ## gke ingress - # kubernetes.io/ingress.class: gce - ## use the below annotation to attach gke managed certificate - ## https://cloud.google.com/kubernetes-engine/docs/how-to/managed-certs#gcloud - # networking.gke.io/managed-certificates: - - ## aws alb annotations - ## aws load balancer controller needs to be installed on the cluster for these annotations to work - ## documentation aws load balancer controller https://kubernetes-sigs.github.io/aws-load-balancer-controller/v2.4/guide/ingress/annotations/ - # alb.ingress.kubernetes.io/actions.ssl-redirect: '{"Type": "redirect", "RedirectConfig": { "Protocol": "HTTPS", "Port": "", "StatusCode": "HTTP_301"}}' - # alb.ingress.kubernetes.io/backend-protocol: HTTPS - ## arn of the certificate available on aws certificate manager - # alb.ingress.kubernetes.io/certificate-arn: "" - # alb.ingress.kubernetes.io/listen-ports: '[{"HTTPS":443}, {"HTTP":80}]' - # alb.ingress.kubernetes.io/scheme: internet-facing - # alb.ingress.kubernetes.io/target-group-attributes: stickiness.enabled=true,stickiness.lb_cookie.duration_seconds=3600 - # alb.ingress.kubernetes.io/target-type: ip diff --git a/helm-overrides/k8s-sec-admin-ase1/external-secrets/custom-values.yaml b/helm-overrides/k8s-sec-admin-ase1/external-secrets/custom-values.yaml deleted file mode 100644 index b7ca117..0000000 --- a/helm-overrides/k8s-sec-admin-ase1/external-secrets/custom-values.yaml +++ /dev/null @@ -1,45 +0,0 @@ -external-secrets: - replicaCount: 1 - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Additional service annotations - annotations: {} - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: devops - nodeSelector: - dedicated: devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: devops - nodeSelector: - dedicated: devops - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: devops - nodeSelector: - dedicated: devops diff --git a/helm-overrides/k8s-sec-admin-ase1/flagger/custom-values.yaml b/helm-overrides/k8s-sec-admin-ase1/flagger/custom-values.yaml deleted file mode 100644 index c703ab2..0000000 --- a/helm-overrides/k8s-sec-admin-ase1/flagger/custom-values.yaml +++ /dev/null @@ -1,54 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: hpa-changes-26 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "512Mi" - cpu: "1000m" - requests: - memory: "256Mi" - cpu: "100m" - -nodeSelector: - dedicated: devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: sec - bu: infra diff --git a/helm-overrides/k8s-sec-admin-ase1/ingress-nginx/custom-values.yaml b/helm-overrides/k8s-sec-admin-ase1/ingress-nginx/custom-values.yaml deleted file mode 100644 index e72502e..0000000 --- a/helm-overrides/k8s-sec-admin-ase1/ingress-nginx/custom-values.yaml +++ /dev/null @@ -1,15 +0,0 @@ -ingress-nginx: - controller: - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-sec-admin"}}}' - nodeSelector: - dedicated: devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-sec-admin-ase1/keda/custom-values.yaml b/helm-overrides/k8s-sec-admin-ase1/keda/custom-values.yaml deleted file mode 100644 index e24a295..0000000 --- a/helm-overrides/k8s-sec-admin-ase1/keda/custom-values.yaml +++ /dev/null @@ -1,18 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" - podLabels: - bu: "infra" - team: "sec" - metricsAdapter: - bu: "infra" - team: "sec" diff --git a/helm-overrides/k8s-sec-admin-ase1/kube-dns/custom-values.yaml b/helm-overrides/k8s-sec-admin-ase1/kube-dns/custom-values.yaml deleted file mode 100644 index c3c32d1..0000000 --- a/helm-overrides/k8s-sec-admin-ase1/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"]} \ No newline at end of file diff --git a/helm-overrides/k8s-sec-admin-ase1/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-sec-admin-ase1/kube-state-metrics/custom-values.yaml deleted file mode 100644 index a5c9fa6..0000000 --- a/helm-overrides/k8s-sec-admin-ase1/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-infra-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-infra-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "infra" - team: "sre" - service: "kube-state-metrics-infra-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "devops" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 10m - memory: 50Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-sec-admin-ase1/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-sec-admin-ase1/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index a54a84e..0000000 --- a/helm-overrides/k8s-sec-admin-ase1/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,491 +0,0 @@ -# # Default values for prometheus-node-exporter. -# # This is a YAML-formatted file. -# # Declare variables to be passed into your templates. -# image: -# registry: quay.io -# repository: prometheus/node-exporter -# # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} -# tag: "" -# pullPolicy: IfNotPresent -# digest: "" - -# imagePullSecrets: [] -# # - name: "image-pull-secret" -# nameOverride: "" -# fullnameOverride: "" - -# # Number of old history to retain to allow rollback -# # Default Kubernetes value is set to 10 -# revisionHistoryLimit: 10 - -# global: -# # To help compatibility with other charts which use global.imagePullSecrets. -# # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). -# # global: -# # imagePullSecrets: -# # - name: pullSecret1 -# # - name: pullSecret2 -# # or -# # global: -# # imagePullSecrets: -# # - pullSecret1 -# # - pullSecret2 -# imagePullSecrets: [] -# # -# # Allow parent charts to override registry hostname -# imageRegistry: "" - -# # Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# # The requests are served through the same service but requests are HTTPS. -# kubeRBACProxy: -# enabled: false -# image: -# registry: quay.io -# repository: brancz/kube-rbac-proxy -# tag: v0.14.0 -# sha: "" -# pullPolicy: IfNotPresent - -# # List of additional cli arguments to configure kube-rbac-prxy -# # for example: --tls-cipher-suites, --log-file, etc. -# # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage -# extraArgs: [] - -# ## Specify security settings for a Container -# ## Allows overrides and additional options compared to (Pod) securityContext -# ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -# containerSecurityContext: {} - -# resources: {} -# # We usually recommend not to specify default resources and to leave this as a conscious -# # choice for the user. This also increases chances charts run on environments with little -# # resources, such as Minikube. If you do want to specify resources, uncomment the following -# # lines, adjust them as necessary, and remove the curly braces after 'resources:'. -# # limits: -# # cpu: 100m -# # memory: 64Mi -# # requests: -# # cpu: 10m -# # memory: 32Mi - -# service: -# enabled: true -# type: ClusterIP -# port: 9200 -# targetPort: 9200 -# nodePort: -# portName: metrics -# listenOnAllInterfaces: true -# annotations: -# prometheus.io/scrape: "true" -# ipDualStack: -# enabled: false -# ipFamilies: ["IPv6", "IPv4"] -# ipFamilyPolicy: "PreferDualStack" - -# # Set a NetworkPolicy with: -# # ingress only on service.port -# # no egress permitted -# networkPolicy: -# enabled: false - -# # Additional environment variables that will be passed to the daemonset -# env: {} -# ## env: -# ## VARIABLE: value - -# prometheus: -# monitor: -# enabled: false -# additionalLabels: {} -# namespace: "" - -# jobLabel: "" - -# # List of pod labels to add to node exporter metrics -# # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor -# podTargetLabels: [] - -# scheme: http -# basicAuth: {} -# bearerTokenFile: -# tlsConfig: {} - -# ## proxyUrl: URL of a proxy that should be used for scraping. -# ## -# proxyUrl: "" - -# ## Override serviceMonitor selector -# ## -# selectorOverride: {} - -# ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. -# ## -# attachMetadata: -# node: false - -# relabelings: [] -# metricRelabelings: [] -# interval: "" -# scrapeTimeout: 10s -# ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") -# apiVersion: "" - -# ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. -# ## -# sampleLimit: 0 - -# ## TargetLimit defines a limit on the number of scraped targets that will be accepted. -# ## -# targetLimit: 0 - -# ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. -# ## -# labelLimit: 0 - -# ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. -# ## -# labelNameLengthLimit: 0 - -# ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. -# ## -# labelValueLengthLimit: 0 - -# # PodMonitor defines monitoring for a set of pods. -# # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor -# # Using a PodMonitor may be preferred in some environments where there is very large number -# # of Node Exporter endpoints (1000+) behind a single service. -# # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, -# # the time series resulting from the configuration through PodMonitor may have different labels. -# # For instance, there will not be the service label any longer which might -# # affect PromQL queries selecting that label. -# podMonitor: -# enabled: false -# # Namespace in which to deploy the pod monitor. Defaults to the release namespace. -# namespace: "" -# # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus -# additionalLabels: {} -# # release: kube-prometheus-stack -# # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. -# podTargetLabels: [] -# # apiVersion defaults to monitoring.coreos.com/v1. -# apiVersion: "" -# # Override pod selector to select pod objects. -# selectorOverride: {} -# # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. -# attachMetadata: -# node: false -# # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. -# jobLabel: "" - -# # Scheme/protocol to use for scraping. -# scheme: "http" -# # Path to scrape metrics at. -# path: "/metrics" - -# # BasicAuth allow an endpoint to authenticate over basic authentication. -# # More info: https://prometheus.io/docs/operating/configuration/#endpoint -# basicAuth: {} -# # Secret to mount to read bearer token for scraping targets. -# # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. -# # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core -# bearerTokenSecret: {} -# # TLS configuration to use when scraping the endpoint. -# tlsConfig: {} -# # Authorization section for this endpoint. -# # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization -# authorization: {} -# # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. -# # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 -# oauth2: {} - -# # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. -# proxyUrl: "" -# # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. -# interval: "" -# # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. -# scrapeTimeout: "" -# # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. -# honorTimestamps: true -# # HonorLabels chooses the metric’s labels on collisions with target labels. -# honorLabels: true -# # Whether to enable HTTP2. Default false. -# enableHttp2: "" -# # Drop pods that are not running. (Failed, Succeeded). -# # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase -# filterRunning: "" -# # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. -# followRedirects: "" -# # Optional HTTP URL parameters -# params: {} - -# # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds -# # relabelings for a few standard Kubernetes fields. The original scrape job’s name -# # is available via the __tmp_prometheus_job_name label. -# # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config -# relabelings: [] -# # MetricRelabelConfigs to apply to samples before ingestion. -# metricRelabelings: [] - -# # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. -# sampleLimit: 0 -# # TargetLimit defines a limit on the number of scraped targets that will be accepted. -# targetLimit: 0 -# # Per-scrape limit on number of labels that will be accepted for a sample. -# # Only valid in Prometheus versions 2.27.0 and newer. -# labelLimit: 0 -# # Per-scrape limit on length of labels name that will be accepted for a sample. -# # Only valid in Prometheus versions 2.27.0 and newer. -# labelNameLengthLimit: 0 -# # Per-scrape limit on length of labels value that will be accepted for a sample. -# # Only valid in Prometheus versions 2.27.0 and newer. -# labelValueLengthLimit: 0 - -# ## Customize the updateStrategy if set -# updateStrategy: -# type: RollingUpdate -# rollingUpdate: -# maxUnavailable: 1 - -# resources: -# # We usually recommend not to specify default resources and to leave this as a conscious -# # choice for the user. This also increases chances charts run on environments with little -# # resources, such as Minikube. If you do want to specify resources, uncomment the following -# # lines, adjust them as necessary, and remove the curly braces after 'resources:'. -# limits: -# cpu: 200m -# memory: 50Mi -# requests: -# cpu: 100m -# memory: 30Mi - -# serviceAccount: -# # Specifies whether a ServiceAccount should be created -# create: true -# # The name of the ServiceAccount to use. -# # If not set and create is true, a name is generated using the fullname template -# name: -# annotations: {} -# # annotations: { -# # iam.gke.io/gcp-service-account: node-exporter-admin-prd@meesho-admin-prd-0622.iam.gserviceaccount.com -# # } -# imagePullSecrets: [] -# automountServiceAccountToken: false - -# securityContext: -# fsGroup: 65534 -# runAsGroup: 65534 -# runAsNonRoot: true -# runAsUser: 65534 - -# containerSecurityContext: -# readOnlyRootFilesystem: true -# # capabilities: -# # add: -# # - SYS_TIME - -# rbac: -# ## If true, create & use RBAC resources -# ## -# create: true -# ## If true, create & use Pod Security Policy resources -# ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -# pspEnabled: true -# pspAnnotations: {} - -# # for deployments that have node_exporter deployed outside of the cluster, list -# # their addresses here -# endpoints: [] - -# # Expose the service to the host network -# hostNetwork: true - -# # Share the host process ID namespace -# hostPID: true - -# # Mount the node's root file system (/) at /host/root in the container -# hostRootFsMount: -# enabled: true -# # Defines how new mounts in existing mounts on the node or in the container -# # are propagated to the container or node, respectively. Possible values are -# # None, HostToContainer, and Bidirectional. If this field is omitted, then -# # None is used. More information on: -# # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation -# mountPropagation: HostToContainer - -# ## Assign a group of affinity scheduling rules -# ## -# affinity: {} -# # nodeAffinity: -# # requiredDuringSchedulingIgnoredDuringExecution: -# # nodeSelectorTerms: -# # - matchFields: -# # - key: metadata.name -# # operator: In -# # values: -# # - target-host-name - -# # Annotations to be added to node exporter pods -# podAnnotations: -# # Fix for very slow GKE cluster upgrades -# cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - -# # Extra labels to be added to node exporter pods -# podLabels: -# bu: "infra" -# team: "infra-sre" -# service: "node-exporter-infra-prd" -# env: "prd" -# priority: "p0" -# type: "exporter" - -# # Annotations to be added to node exporter daemonset -# daemonsetAnnotations: -# prometheus.io/scrape: "true" -# prometheus.io/port: "9200" -# prometheus.io/path: "/metrics" - -# ## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -# releaseLabel: false - -# # Custom DNS configuration to be added to prometheus-node-exporter pods -# dnsConfig: {} -# # nameservers: -# # - 1.2.3.4 -# # searches: -# # - ns1.svc.cluster-domain.example -# # - my.dns.search.suffix -# # options: -# # - name: ndots -# # value: "2" -# # - name: edns0 - -# ## Assign a nodeSelector if operating a hybrid cluster -# ## -# nodeSelector: {} -# # kubernetes.io/os: linux -# # kubernetes.io/arch: amd64 - -# tolerations: -# - operator: Exists - -# ## Assign a PriorityClassName to pods if set -# # priorityClassName: "" - -# ## Additional container arguments -# ## -# extraArgs: [] -# # - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# # - --collector.textfile.directory=/run/prometheus - -# ## Additional mounts from the host to node-exporter container -# ## -# extraHostVolumeMounts: [] -# # - name: -# # hostPath: -# # mountPath: -# # readOnly: true|false -# # mountPropagation: None|HostToContainer|Bidirectional - -# ## Additional configmaps to be mounted. -# ## -# configmaps: [] -# # - name: -# # mountPath: -# secrets: [] -# # - name: -# # mountPath: -# ## Override the deployment namespace -# ## -# namespaceOverride: "monitoring" - -# ## Additional containers for export metrics to text file -# ## -# sidecars: [] -# ## - name: nvidia-dcgm-exporter -# ## image: nvidia/dcgm-exporter:1.4.3 - -# ## Volume for sidecar containers -# ## -# sidecarVolumeMount: [] -# ## - name: collector-textfiles -# ## mountPath: /run/prometheus -# ## readOnly: false - -# ## Additional mounts from the host to sidecar containers -# ## -# sidecarHostVolumeMounts: [] -# # - name: -# # hostPath: -# # mountPath: -# # readOnly: true|false -# # mountPropagation: None|HostToContainer|Bidirectional - -# ## Additional InitContainers to initialize the pod -# ## -# extraInitContainers: [] - -# ## Liveness probe -# ## -# livenessProbe: -# failureThreshold: 3 -# httpGet: -# httpHeaders: [] -# scheme: http -# initialDelaySeconds: 0 -# periodSeconds: 10 -# successThreshold: 1 -# timeoutSeconds: 1 - -# ## Readiness probe -# ## -# readinessProbe: -# failureThreshold: 3 -# httpGet: -# httpHeaders: [] -# scheme: http -# initialDelaySeconds: 0 -# periodSeconds: 10 -# successThreshold: 1 -# timeoutSeconds: 1 - -# # Enable vertical pod autoscaler support for prometheus-node-exporter -# verticalPodAutoscaler: -# enabled: false - -# # Recommender responsible for generating recommendation for the object. -# # List should be empty (then the default recommender will generate the recommendation) -# # or contain exactly one recommender. -# # recommenders: -# # - name: custom-recommender-performance - -# # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory -# controlledResources: [] -# # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. -# # controlledValues: RequestsAndLimits - -# # Define the max allowed resources for the pod -# maxAllowed: {} -# # cpu: 200m -# # memory: 100Mi -# # Define the min allowed resources for the pod -# minAllowed: {} -# # cpu: 200m -# # memory: 100Mi - -# # updatePolicy: -# # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction -# # minReplicas: 1 -# # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates -# # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". -# # updateMode: Auto - -# # Extra manifests to deploy as an array -# extraManifests: [] -# # - | -# # apiVersion: v1 -# # kind: ConfigMap -# # metadata: -# # name: prometheus-extra -# # data: -# # extra-data: "value" diff --git a/helm-overrides/k8s-sec-admin-ase1/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/k8s-sec-admin-ase1/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index 4b845f8..0000000 --- a/helm-overrides/k8s-sec-admin-ase1/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# # Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -# nameOverride: "stackdriver-exporter-infra-prd" - -# # Provide a name to substitute for the full names of resources -# fullnameOverride: "stackdriver-exporter-infra-prd" - -# # Number of exporters to run -# replicaCount: 1 - -# # Restart policy for container -# restartPolicy: Always - -# image: -# repository: prometheuscommunity/stackdriver-exporter -# # if not set appVersion field from Chart.yaml is used -# tag: "" -# pullPolicy: IfNotPresent - -# ## Optionally specify an array of imagePullSecrets. -# ## Secrets must be manually created in the namespace. -# ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -# ## -# # pullSecrets: -# # - myDockerConfigJsonSecretName - -# resources: -# requests: -# cpu: 500m -# memory: 512Mi -# # limits: -# # cpu: 100m -# # memory: 128Mi - -# securityContext: {} - -# containerSecurityContext: {} - -# service: -# type: ClusterIP -# httpPort: 9255 -# annotations: {} - -# ## Additional labels to add to all resources -# customLabels: -# bu: "infra" -# team: "infra-sre" -# service: "stackdriver-exporter-infra-prd" -# env: "prd" -# priority: "p0" -# type: "exporter" -# # app: prometheus-stackdriver-exporter - -# secret: -# labels: {} - -# stackdriver: -# # The Google Project ID to gather metrics for -# projectId: "meesho-admin-prd-0622" -# # An existing secret which contains credentials.json -# serviceAccountSecret: "" -# # Provide custom key for the existing secret to load credentials.json from -# serviceAccountSecretKey: "" -# # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account -# serviceAccountKey: "" -# # Max number of retries that should be attempted on 503 errors from Stackdriver -# maxRetries: 0 -# # How long should Stackdriver_exporter wait for a result from the Stackdriver API -# httpTimeout: 10s -# # Max time between each request in an exp backoff scenario -# maxBackoff: 5s -# # The amount of jitter to introduce in an exp backoff scenario -# backoffJitter: 1s -# # The HTTP statuses that should trigger a retry -# retryStatuses: 503 -# # Drop metrics from attached projects and fetch `project_id` only -# dropDelegatedProjects: false -# metrics: -# # The prefixes to gather metrics for, we default to just CPU metrics. -# typePrefixes: 'compute.googleapis.com/instance/cpu' -# # The filters to refine the metrics query by using Filter objects that Google provides. -# # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] -# # https://cloud.google.com/monitoring/api/v3/filters -# filters: [] -# # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' -# # The frequency to request -# interval: '5m' -# # How far into the past to offset -# offset: '0s' -# # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. -# ingestDelay: false -# # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. -# aggregateDeltas: false -# # How long should a delta metric continue to be exported after GCP stops producing a metric -# aggregateDeltasTTL: '30m' - -# web: -# # Port to listen on -# listenAddress: ':9255' -# # Path under which to expose metrics. -# path: /metrics - -# ## Pod affinity -# ## -# affinity: {} - -# annotations: -# prometheus.io/scrape: "true" -# prometheus.io/port: "9255" -# prometheus.io/path: "/metrics" - -# ## Pod extra arguments -# ## -# extraArgs: {} - -# ## Node labels for stackdriver-exporter pod assignment -# ## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -# ## -# nodeSelector: -# dedicated: vmselect - -# ## Node tolerations for stackdriver-exporter scheduling to nodes with taints -# ## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -# ## -# tolerations: -# - key: "dedicated" -# operator: "Equal" -# value: "vmselect" -# effect: "NoSchedule" - - -# ## Service Account -# ## -# serviceAccount: -# # Specifies whether a ServiceAccount should be created -# create: true -# # The name of the ServiceAccount to use. -# # If not set and create is false, 'default' is used -# # If not set and create is true, a name is generated using the fullname template -# name: -# annotations: { -# iam.gke.io/gcp-service-account: sa-stackdriver-exp-infra-prd@meesho-admin-prd-0622.iam.gserviceaccount.com -# } - - -# # Enable this if you're using https://github.com/coreos/prometheus-operator -# serviceMonitor: -# enabled: false -# namespace: monitoring -# # additionalLabels is the set of additional labels to add to the ServiceMonitor -# additionalLabels: {} -# # How long until a scrape request times out. -# scrapeTimeout: '10s' -# # fallback to the prometheus default unless specified -# interval: 10s -# # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) -# honorLabels: true -# # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter -# honorTimestamps: true -# # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig -# metricRelabelings: [] -# # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig -# relabelings: [] - -# ## Custom PrometheusRules to be defined -# ## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -# prometheusRule: -# enabled: false -# additionalLabels: {} -# namespace: "" -# rules: [] diff --git a/helm-overrides/k8s-sec-admin-ase1/telegraf-operator/custom-values.yaml b/helm-overrides/k8s-sec-admin-ase1/telegraf-operator/custom-values.yaml deleted file mode 100644 index 5dce0f9..0000000 --- a/helm-overrides/k8s-sec-admin-ase1/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,152 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [global_tags] - priority_v2 = "$PRIORITY_V2" - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "sec-admin" - team: "sec-admin-sre" - service: "telegraf-operator-sec-admin-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - dedicated: "vmselect" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false \ No newline at end of file diff --git a/helm-overrides/k8s-sec-admin-ase1/victoria-metrics-agent/custom-values.yaml b/helm-overrides/k8s-sec-admin-ase1/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index 9b532c0..0000000 --- a/helm-overrides/k8s-sec-admin-ase1/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,295 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-sec-admin -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-sec-admin-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-demand.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-sec-admin.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "infra" - team: "sec" - service: "vmagent-sec-admin" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-sec-admin"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: metricsapi-sec-admin.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 8 - memory: 8Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "devops" - -tolerations: - - key: "dedicated" - operator: "Equal" - #value: "vmagent" - value: "devops" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-sec-admin-ase1/victoria-metrics-insert/custom-values.yaml b/helm-overrides/k8s-sec-admin-ase1/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index 0a2fe49..0000000 --- a/helm-overrides/k8s-sec-admin-ase1/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,228 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-sec-admin - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-sec-admin - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "infra" - team: "sec" - service: "vminsert-sec-admin" - env: "prd" - priority: "p0" - type: "vminsert" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 2 - # -- Metric for HPA to use to scale the vminsert component - metrics: [] - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "devops" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 4 - memory: 2Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-sec-admin.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-sec-admin-ase1/victoria-metrics-select/custom-values.yaml b/helm-overrides/k8s-sec-admin-ase1/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index 0c31b07..0000000 --- a/helm-overrides/k8s-sec-admin-ase1/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,290 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-sec-admin - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-sec-admin - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 180s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "infra" - team: "sec" - service: "vmselect-sec-admin" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 50 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "devops" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 13 - memory: 40Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-sec-admin.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-sec-admin-ase1/victoria-metrics-storage/custom-values.yaml b/helm-overrides/k8s-sec-admin-ase1/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index 26c94d9..0000000 --- a/helm-overrides/k8s-sec-admin-ase1/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,317 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-sec-admin - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "devops" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "devops" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - #storageClass: pd-standard-retain - storageClass: standard - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 700Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "infra" - team: "sec" - service: "vmstorage-sec-admin" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 5 - memory: 90Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.93.7-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-demand-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-shared-int-ase1/bifrost/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/bifrost/custom-values.yaml deleted file mode 100644 index e7e5118..0000000 --- a/helm-overrides/k8s-shared-int-ase1/bifrost/custom-values.yaml +++ /dev/null @@ -1,281 +0,0 @@ -# Custom values for Bifrost (ai-gateway) - Meesho Production -# Usage: helm install bifrost ./helm-templates/bifrost/ -f ./helm-templates/bifrost/custom-values.yaml -n int-ai-gateway - -# -- Deployment Configuration -- -replicaCount: 2 - -fullnameOverride: "int-ai-gateway" - -image: - repository: docker.io/maximhq/bifrost - pullPolicy: IfNotPresent - tag: "v1.4.7" - -# -- Service Account -- -serviceAccount: - create: true - automount: true - annotations: - iam.gke.io/gcp-service-account: sa-dvops-ai-gateway@meesho-central-prd-0622.iam.gserviceaccount.com - name: "int-ai-gateway" - -# -- Pod Metadata -- -deploymentLabels: - bu: central - env: int - team: devops - priority: p1 - priority_v2: sp1 - primary_owner: deep.shah - secondary_owner: mahak.jain - service: ai-gateway - service_type: producer-httpstateless - -podLabels: - bu: central - env: int - team: devops - priority: p1 - priority_v2: sp1 - primary_owner: deep.shah - secondary_owner: mahak.jain - service: ai-gateway - service_type: producer-httpstateless - -podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: "8080" - prometheus.io/scrape: "true" - telegraf.influxdata.com/class: infra - -# -- Security Context -- -podSecurityContext: - fsGroup: 65534 - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - -securityContext: - capabilities: - drop: - - ALL - readOnlyRootFilesystem: false - runAsNonRoot: true - runAsUser: 65534 - -# -- Service -- -service: - type: ClusterIP - port: 8080 - -# -- Contour HTTPProxy -- -# ingress.enabled=false disables Bifrost's official K8s Ingress -# httpProxy.enabled=true enables the Meesho Contour HTTPProxy templates -# HTTPProxy templates read from ingress.* for hosts, class, etc. -httpProxy: - enabled: true -createContourGateway: true -namespace: int-ai-gateway -contourResponseTimeout: false -ingress: - enabled: false - ingressClassName: contour-internal-1 - servicePortNumber: 8080 - enableWebsocket: false - hosts: - - host: ai-gateway.int.meesho.int - paths: - - path: / - pathType: ImplementationSpecific - slowStart: - enabled: false - aggression: 1 - minPercent: 10 - window: 120s - -# -- Resources -- -resources: - limits: - cpu: "1" - memory: 2Gi - requests: - cpu: "500m" - memory: 1Gi - -# -- Health Probes -- -livenessProbe: - httpGet: - path: /health - port: http - scheme: HTTP - initialDelaySeconds: 15 - periodSeconds: 5 - timeoutSeconds: 3 - failureThreshold: 5 - successThreshold: 1 - -readinessProbe: - httpGet: - path: /health - port: http - scheme: HTTP - initialDelaySeconds: 15 - periodSeconds: 5 - timeoutSeconds: 3 - failureThreshold: 5 - successThreshold: 1 - -# -- HPA (disabled - using KEDA) -- -autoscaling: - enabled: false - -# -- Scheduling -- -nodeSelector: - cloud.google.com/compute-class: preprod-cost-optimized - -tolerations: - - key: cloud.google.com/compute-class - operator: Equal - value: preprod-cost-optimized - effect: NoSchedule - -affinity: {} - -# -- Lifecycle & Graceful Shutdown -- -terminationGracePeriodSeconds: 300 -lifecycle: - preStop: - exec: - command: - - /bin/bash - - "-c" - - "kill -SIGQUIT; /bin/sleep 120" - -# -- Bifrost Application Config -- -bifrost: - appDir: /app/data - port: 8080 - host: 0.0.0.0 - logLevel: warn - logStyle: json - - # Auth configured via Bifrost UI (stored in DB), not in Helm values - # This avoids blocking /metrics scrape while still protecting the dashboard - - client: - dropExcessRequests: false - initialPoolSize: 300 - allowedOrigins: - - "*" - enableLogging: true - disableContentLogging: false - disableDbPingsInHealth: false - logRetentionDays: 365 - enforceGovernanceHeader: false - allowDirectKeys: false - maxRequestBodySizeMb: 100 - enableLitellmFallbacks: false - - # Configure providers with env.VAR_NAME references for API keys - # providers: - # openai: - # - keys: - # - value: "env.OPENAI_API_KEY" - # models: ["gpt-4o", "gpt-4o-mini"] - # weight: 1.0 - -# -- Storage (External PostgreSQL) -- -storage: - mode: postgres - configStore: - enabled: true - logsStore: - enabled: true - -postgresql: - enabled: false - external: - enabled: true - host: "10.224.128.143" - port: 5432 - user: "ai-gateway" - database: "bifrost_db_4_7" - sslMode: "disable" - existingSecret: "int-ai-gateway-vault" - passwordKey: "BIFROST_ai_gateway" - -# -- Vector Store (disabled) -- -vectorStore: - enabled: false - type: none - -# -- Meesho Standard Env Vars -- -env: - - name: TZ - value: "Asia/Kolkata" - - name: TELEGRAF_UDP_HOST - valueFrom: - fieldRef: - fieldPath: status.podIP - - name: NODE_IP - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - name: POD_NAME - valueFrom: - fieldRef: - fieldPath: metadata.name - - name: POD_NAMESPACE - valueFrom: - fieldRef: - fieldPath: metadata.namespace - - name: POD_IP - valueFrom: - fieldRef: - fieldPath: status.podIP - -# --- Meesho Infrastructure Extensions --- - -# -- PodDisruptionBudget -- -podDisruptionBudget: - enabled: true - maxUnavailable: "10%" - -# -- ExternalSecret (Vault) -- -# Creates K8s Secret "int-ai-gateway-vault" from Vault path -# This secret is referenced by postgresql.external.existingSecret above -externalSecret: - enabled: true - secretName: "int-ai-gateway-vault" - path: "int/cntr/devop/ai-gateway" - refreshInterval: "0" - secretStoreRef: "vault-backend" - -# -- KEDA ScaledObject -- -keda: - enabled: true - pollingInterval: 30 - minReplicaCount: 2 - maxReplicaCount: 200 - scaledown: - stabilizationWindowSeconds: 1800 - selectpolicy: Min - policies: - - type: Pods - value: 2 - periodseconds: 15 - scaleup: - stabilizationWindowSeconds: 120 - selectpolicy: Max - policies: - - type: Pods - value: 2 - periodseconds: 15 - - type: Percent - value: 10 - periodseconds: 15 - triggers: - - type: cpu - metricType: Utilization - metadata: - value: "40" diff --git a/helm-overrides/k8s-shared-int-ase1/cert-manager/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/cert-manager/custom-values.yaml deleted file mode 100644 index c4df99f..0000000 --- a/helm-overrides/k8s-shared-int-ase1/cert-manager/custom-values.yaml +++ /dev/null @@ -1,129 +0,0 @@ -global: - logLevel: 2 - rbac: - create: true - priorityClassName: "high-priority" -installCRDs: false - -crds: - enabled: true - keep: true - -# Cert-manager Controller -replicaCount: 1 -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-controller - tag: v1.20.1 - pullPolicy: IfNotPresent - -nodeSelector: - cloud.google.com/compute-class: cc-contour-vmagent-od - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "cc-contour-vmagent-od" - effect: "NoSchedule" - -resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 500m - memory: 512Mi - -# Webhook Configuration -webhook: - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-webhook - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: cc-contour-vmagent-od - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "cc-contour-vmagent-od" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# CA Injector Configuration -cainjector: - enabled: true - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-cainjector - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: cc-contour-vmagent-od - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "cc-contour-vmagent-od" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# ACME Solver Configuration -acmesolver: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-acmesolver - tag: v1.20.1 - pullPolicy: IfNotPresent - -# Startup API Check -startupapicheck: - enabled: true - timeout: 1m - backoffLimit: 4 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-startupapicheck - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - cloud.google.com/compute-class: cc-contour-vmagent-od - - tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "cc-contour-vmagent-od" - effect: "NoSchedule" - - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - -# Prometheus Monitoring -prometheus: - enabled: true - servicemonitor: - enabled: false - interval: 60s - scrapeTimeout: 30s - labels: - prometheus: cert-manager diff --git a/helm-overrides/k8s-shared-int-ase1/compute-class/cc-contour-vmagent-od-compute-class.yaml b/helm-overrides/k8s-shared-int-ase1/compute-class/cc-contour-vmagent-od-compute-class.yaml deleted file mode 100644 index 992f2fb..0000000 --- a/helm-overrides/k8s-shared-int-ase1/compute-class/cc-contour-vmagent-od-compute-class.yaml +++ /dev/null @@ -1,21 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: cc-contour-vmagent-od -spec: - nodePoolConfig: - serviceAccount: sa-shr-xshr-od-amd-int@meesho-shared-int-0525.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - priorities: - - machineType: n2d-highcpu-8 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp - nodePoolAutoCreation: - enabled: true - activeMigration: - optimizeRulePriority: false diff --git a/helm-overrides/k8s-shared-int-ase1/compute-class/cc-shared-cost-optimised-compute-class.yaml b/helm-overrides/k8s-shared-int-ase1/compute-class/cc-shared-cost-optimised-compute-class.yaml deleted file mode 100644 index 79862c5..0000000 --- a/helm-overrides/k8s-shared-int-ase1/compute-class/cc-shared-cost-optimised-compute-class.yaml +++ /dev/null @@ -1,55 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: preprod-cost-optimized -spec: - nodePoolConfig: - serviceAccount: sa-shr-xshr-od-amd-int@meesho-shared-int-0525.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - priorities: - - machineType: n4d-standard-16 - maxPodsPerNode: 32 - spot: true - storage: - bootDiskSize: 80 - bootDiskType: hyperdisk-balanced - - machineType: n4d-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 80 - bootDiskType: hyperdisk-balanced - - machineType: n2d-standard-16 - maxPodsPerNode: 32 - spot: true - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - - machineType: n2-standard-16 - maxPodsPerNode: 32 - spot: true - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - - machineType: n2d-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - - machineType: n2-standard-16 - maxPodsPerNode: 32 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: true - autoscalingPolicy: - consolidationDelayMinutes: 3 - consolidationThreshold: 50 - whenUnsatisfiable: DoNotScaleUp diff --git a/helm-overrides/k8s-shared-int-ase1/compute-class/cc-shared-devops-od-compute-class.yaml b/helm-overrides/k8s-shared-int-ase1/compute-class/cc-shared-devops-od-compute-class.yaml deleted file mode 100644 index 4392da9..0000000 --- a/helm-overrides/k8s-shared-int-ase1/compute-class/cc-shared-devops-od-compute-class.yaml +++ /dev/null @@ -1,21 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: cc-shared-devops-od -spec: - nodePoolConfig: - serviceAccount: sa-shr-xshr-od-amd-int@meesho-shared-int-0525.iam.gserviceaccount.com - priorityDefaults: - location: - zones: ['asia-southeast1-a'] - priorities: - - machineType: e2-standard-4 - spot: false - storage: - bootDiskSize: 30 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp - nodePoolAutoCreation: - enabled: true - activeMigration: - optimizeRulePriority: false diff --git a/helm-overrides/k8s-shared-int-ase1/compute-class/cc-shared-gpu-a2-compute-class.yaml b/helm-overrides/k8s-shared-int-ase1/compute-class/cc-shared-gpu-a2-compute-class.yaml deleted file mode 100644 index d16bec7..0000000 --- a/helm-overrides/k8s-shared-int-ase1/compute-class/cc-shared-gpu-a2-compute-class.yaml +++ /dev/null @@ -1,25 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: preprod-gpu-a2-priority -spec: - priorities: - - machineFamily: a2 - gpu: - type: nvidia-tesla-a100 - count: 1 - driverVersion: default - spot: true - - machineFamily: a2 - gpu: - type: nvidia-tesla-a100 - count: 1 - driverVersion: default - spot: false - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: false - autoscalingPolicy: - consolidationDelayMinutes: 3 - consolidationThreshold: 50 \ No newline at end of file diff --git a/helm-overrides/k8s-shared-int-ase1/compute-class/cc-shared-gpu-a3-compute-class.yaml b/helm-overrides/k8s-shared-int-ase1/compute-class/cc-shared-gpu-a3-compute-class.yaml deleted file mode 100644 index 709f278..0000000 --- a/helm-overrides/k8s-shared-int-ase1/compute-class/cc-shared-gpu-a3-compute-class.yaml +++ /dev/null @@ -1,25 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: preprod-gpu-a3-priority -spec: - priorities: - - machineFamily: a3 - gpu: - type: nvidia-h100-80gb - count: 1 - driverVersion: default - spot: true - - machineFamily: a3 - gpu: - type: nvidia-h100-80gb - count: 1 - driverVersion: default - spot: false - activeMigration: - optimizeRulePriority: true - nodePoolAutoCreation: - enabled: false - autoscalingPolicy: - consolidationDelayMinutes: 3 - consolidationThreshold: 50 \ No newline at end of file diff --git a/helm-overrides/k8s-shared-int-ase1/compute-class/cc-starrocks-n2-highmem-16.yaml b/helm-overrides/k8s-shared-int-ase1/compute-class/cc-starrocks-n2-highmem-16.yaml deleted file mode 100644 index 84fd2f2..0000000 --- a/helm-overrides/k8s-shared-int-ase1/compute-class/cc-starrocks-n2-highmem-16.yaml +++ /dev/null @@ -1,16 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: int-starrocks-n2-highmem-16 -spec: - priorities: - - machineType: n2-highmem-16 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp - nodePoolAutoCreation: - enabled: true - activeMigration: - optimizeRulePriority: false diff --git a/helm-overrides/k8s-shared-int-ase1/compute-class/cc-starrocks-n2-highmem-32.yaml b/helm-overrides/k8s-shared-int-ase1/compute-class/cc-starrocks-n2-highmem-32.yaml deleted file mode 100644 index 430a26a..0000000 --- a/helm-overrides/k8s-shared-int-ase1/compute-class/cc-starrocks-n2-highmem-32.yaml +++ /dev/null @@ -1,16 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: int-starrocks-n2-highmem-32 -spec: - priorities: - - machineType: n2-highmem-32 - spot: false - storage: - bootDiskSize: 120 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp - nodePoolAutoCreation: - enabled: true - activeMigration: - optimizeRulePriority: false diff --git a/helm-overrides/k8s-shared-int-ase1/compute-class/cc-starrocks-n2-highmem-8.yaml b/helm-overrides/k8s-shared-int-ase1/compute-class/cc-starrocks-n2-highmem-8.yaml deleted file mode 100644 index 3cbe078..0000000 --- a/helm-overrides/k8s-shared-int-ase1/compute-class/cc-starrocks-n2-highmem-8.yaml +++ /dev/null @@ -1,16 +0,0 @@ -apiVersion: cloud.google.com/v1 -kind: ComputeClass -metadata: - name: int-starrocks-n2-highmem-8 -spec: - priorities: - - machineType: n2-highmem-8 - spot: false - storage: - bootDiskSize: 100 - bootDiskType: pd-ssd - whenUnsatisfiable: DoNotScaleUp - nodePoolAutoCreation: - enabled: true - activeMigration: - optimizeRulePriority: false diff --git a/helm-overrides/k8s-shared-int-ase1/conntrack-adjuster/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index b131f7b..0000000 --- a/helm-overrides/k8s-shared-int-ase1/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,25 +0,0 @@ -daemonSet: - namespace: int-conntrack-adjuster - -conntrack: - # Maximum number of conntrack entries - max: 2097152 - # Hash size for conntrack - hashsize: 524288 - # Sleep interval between adjustments (seconds) - sleepInterval: 30 - -# Enable additional matchExpressions -addExtraMatchExpressions: true - -# Additional matchExpressions to append -additionalMatchExpressions: -- key: cloud.google.com/compute-class - operator: In - values: - - "cc-contour-vmagent-od" -# - key: environment -# operator: In -# values: -# - "production" -# - "staging" diff --git a/helm-overrides/k8s-shared-int-ase1/contour-ca-issuer/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/contour-ca-issuer/custom-values.yaml deleted file mode 100644 index 9301c04..0000000 --- a/helm-overrides/k8s-shared-int-ase1/contour-ca-issuer/custom-values.yaml +++ /dev/null @@ -1,17 +0,0 @@ -# ClusterIssuer override for k8s-shared-int-ase1 (pre-prod shared cluster). -# ClusterIssuer is a cluster-scoped resource — this file pins the issuer -# identity for this cluster so the name is auditable per-cluster. -# -# Must match the `issuerRef.name` in consuming Certificate CRs (see -# devops-helm-charts/2.0.0/templates/proxyless-grpc-cert.yaml). - -issuerName: contour-shared-int-ca-issuer -rootCASecretName: contour-shared-ca - -externalSecret: - enabled: true - vaultPath: meesho/devops/contour/root-ca - refreshInterval: "0" - secretStoreRef: - name: vault-backend -namespace: cert-manager-shared-int \ No newline at end of file diff --git a/helm-overrides/k8s-shared-int-ase1/contour-cert-checker/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/contour-cert-checker/custom-values.yaml deleted file mode 100644 index 77a27be..0000000 --- a/helm-overrides/k8s-shared-int-ase1/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,37 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=k8s-shared-int-ase1"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - cloud.google.com/compute-class: cc-shared-devops-od - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: cc-shared-devops-od - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/k8s-shared-int-ase1/contour-external/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/contour-external/custom-values.yaml deleted file mode 100644 index 47e39cc..0000000 --- a/helm-overrides/k8s-shared-int-ase1/contour-external/custom-values.yaml +++ /dev/null @@ -1,115 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - network: - num-trusted-hops: 1 -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: shared - team: shared-devops - env: int - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: kubernetes.io/arch - operator: Equal - value: arm64 - - effect: NoSchedule - key: dedicated - operator: Equal - value: n4a-contour - - nodeSelector: - dedicated: n4a-contour - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: shared - team: shared-devops - env: int - kind: deployment - tolerations: - - effect: NoSchedule - key: kubernetes.io/arch - operator: Equal - value: arm64 - - effect: NoSchedule - key: dedicated - operator: Equal - value: n4a-contour - - nodeSelector: - dedicated: n4a-contour - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 1000m - memory: 512Mi - service: - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-shared-int"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-shared-int-ase1/contour-internal-0/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/contour-internal-0/custom-values.yaml deleted file mode 100644 index 8b84dba..0000000 --- a/helm-overrides/k8s-shared-int-ase1/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,125 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 1 - podLabels: - bu: shared - team: shared-devops - env: int - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: kubernetes.io/arch - operator: Equal - value: arm64 - - effect: NoSchedule - key: dedicated - operator: Equal - value: n4a-contour - - nodeSelector: - dedicated: n4a-contour - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - updateStrategy: - rollingUpdate: - maxSurge: 10% - enabled: true - podLabels: - bu: shared - team: shared-devops - env: int - kind: deployment - tolerations: - - effect: NoSchedule - key: kubernetes.io/arch - operator: Equal - value: arm64 - - effect: NoSchedule - key: dedicated - operator: Equal - value: n4a-contour - - nodeSelector: - dedicated: n4a-contour - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a - logLevel: error - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 350 - targetCPU: "55" - targetMemory: "55" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 1500m - memory: 512Mi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int-0-shared-int"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-shared-int-ase1/contour-internal-1/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/contour-internal-1/custom-values.yaml deleted file mode 100644 index f41f775..0000000 --- a/helm-overrides/k8s-shared-int-ase1/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 1 - podLabels: - bu: shared - team: shared-devops - env: int - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: kubernetes.io/arch - operator: Equal - value: arm64 - - effect: NoSchedule - key: dedicated - operator: Equal - value: n4a-contour - - nodeSelector: - dedicated: n4a-contour - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: shared - team: shared-devops - env: int - kind: deployment - tolerations: - - effect: NoSchedule - key: kubernetes.io/arch - operator: Equal - value: arm64 - - effect: NoSchedule - key: dedicated - operator: Equal - value: n4a-contour - nodeSelector: - dedicated: n4a-contour - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a - logLevel: error - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 350 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 1500m - memory: 512Mi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int-1-shared-int"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-shared-int-ase1/contour-internal-intra-1/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index a66e55b..0000000 --- a/helm-overrides/k8s-shared-int-ase1/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,109 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 1 - podLabels: - bu: shared - team: shared-devops - env: int - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 1024Mi - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: cc-contour-vmagent-od - - nodeSelector: - cloud.google.com/compute-class: cc-contour-vmagent-od - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a - service: - tcpLB: true - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: shared - team: shared-devops - env: int - kind: deployment - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: cc-contour-vmagent-od - nodeSelector: - cloud.google.com/compute-class: cc-contour-vmagent-od - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a - logLevel: error - autoscaling: - enabled: true - minReplicas: 2 - maxReplicas: 350 - targetCPU: "60" - targetMemory: "60" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 1500m - memory: 512Mi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int-intra-1-shared-int"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-shared-int-ase1/coredns/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/coredns/custom-values.yaml deleted file mode 100644 index fe4a0e2..0000000 --- a/helm-overrides/k8s-shared-int-ase1/coredns/custom-values.yaml +++ /dev/null @@ -1,38 +0,0 @@ -replicaCount: 3 - -labels: - bu: shared - team: shared-devops - env: int - -clusterIP: 10.225.0.2 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: cloud.google.com/compute-class - operator: "Equal" - value: "cc-shared-devops-od" - effect: "NoSchedule" - -nodeSelector: - cloud.google.com/compute-class: cc-shared-devops-od - -rewrites: null - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a \ No newline at end of file diff --git a/helm-overrides/k8s-shared-int-ase1/deepgram-onprem/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/deepgram-onprem/custom-values.yaml deleted file mode 100644 index 8d42294..0000000 --- a/helm-overrides/k8s-shared-int-ase1/deepgram-onprem/custom-values.yaml +++ /dev/null @@ -1,897 +0,0 @@ -global: - # -- (string) If using images from the Deepgram Quay image repositories, - # or another private registry to which your cluster doesn't have default access, - # you will need to provide a pre-configured K8s Secret - # with image repository credentials. See chart docs for more details. - pullSecretRef: dg-regcred - - # -- (string) Name of the pre-configured K8s Secret containing your Deepgram - # self-hosted API key. See chart docs for more details. - deepgramSecretRef: dg-self-hosted-api-key - - # -- Additional labels to add to all Deepgram resources - additionalLabels: {} - - # -- When an API or Engine container is signaled to shutdown via Kubernetes sending a SIGTERM - # signal, the container will stop listening on its port, and no new requests will be routed - # to that container. However, the container will continue to run until all existing - # batch or streaming requests have completed, after which it will gracefully shut down. - # - # Batch requests should be finished within 10-15 minutes, but streaming requests can proceed indefinitely. - # - # outstandingRequestGracePeriod defines the period (in sec) after which Kubernetes will forcefully - # shutdown the container, terminating any outstanding connections. 1800 / 60 sec/min = 30 mins - outstandingRequestGracePeriod: 1800 - -# -- Configuration options for horizontal scaling of Deepgram -# services. Only one of `static` and `auto` options can be enabled. -# @default -- `` - -apiAutoscaling: - enabled: true - targetName: deepgram-api - maxReplicas: 30 - minReplicas: 2 - pollingInterval: 30 - scaledown: - policies: - - periodseconds: 60 - type: Pods - value: 2 - selectpolicy: Min - stabilizationWindowSeconds: 900 - scaleup: - policies: - - periodseconds: 15 - type: Pods - value: 2 - - periodseconds: 15 - type: Percent - value: 10 - selectpolicy: Max - stabilizationWindowSeconds: 60 - triggers: - - metadata: - metricName: engine_active_requests - query: sum(kube_deployment_status_replicas{namespace="dg-self-hosted-int",deployment="deepgram-engine"}) - serverAddress: http://vmselect-infra-prd-test-proxy.victoriametrics.svc.clusterset.local:8480/select/100/prometheus/ - threshold: "0.5" - type: prometheus - - - -engineAutoscaling: - enabled: true - targetName: deepgram-engine - maxReplicas: 15 - minReplicas: 2 - pollingInterval: 30 - scaledown: - policies: - - periodseconds: 60 - type: Pods - value: 2 - selectpolicy: Min - stabilizationWindowSeconds: 900 - scaleup: - policies: - - periodseconds: 15 - type: Pods - value: 2 - - periodseconds: 15 - type: Percent - value: 10 - selectpolicy: Max - stabilizationWindowSeconds: 60 - triggers: - - metadata: - metricName: engine_active_requests - query: sum(engine_active_requests{kind="stream",cluster="k8s-shared-int-ase1"}) - serverAddress: http://vmselect-infra-prd-test-proxy.victoriametrics.svc.clusterset.local:8480/select/100/prometheus/ - threshold: "5" - type: prometheus - -scaling: - # -- Number of replicas to set during initial installation. - # @default -- `` - replicas: - api: 1 - engine: 1 - - # -- Enable pod autoscaling based on system load/traffic. - # @default -- `` - auto: - enabled: false - - api: - metrics: - # -- Scale the API deployment to this Engine-to-Api pod ratio - engineToApiRatio: 4 - # -- (list) If you have custom metrics you would like to scale with, you may add them here. - # See the [k8s docs](https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/) - # for how to structure a list of metrics - custom: - - # -- [Configurable scaling behavior](https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#configurable-scaling-behavior) - # @default -- "*See values.yaml file for default*" - behavior: - scaleDown: - policies: - - type: Pods - value: 1 - periodSeconds: 60 - - type: Percent - value: 25 - periodSeconds: 60 - - engine: - # -- Minimum number of Engine replicas. - minReplicas: 1 - # -- Maximum number of Engine replicas. - maxReplicas: 10 - metrics: - # -- If `engine.concurrencyLimit.activeRequests` is set, this variable will - # define the ratio of current active requests to maximum active requests at which - # the Engine pods will scale. Setting this value too close to 1.0 may lead to a situation where - # the cluster is at max capacity and rejects incoming requests. Setting the ratio too close to 0.0 - # will over-optimistically scale your cluster and increase compute costs unnecessarily. - requestCapacityRatio: 0.8 - speechToText: - batch: - # -- (int) Scale the Engine pods based on a static desired number of speech-to-text batch requests per pod - requestsPerPod: 12 - streaming: - # -- (int) Scale the Engine pods based on a static desired number of speech-to-text streaming requests per pod - requestsPerPod: 14 - textToSpeech: - batch: - # -- (int) Scale the Engine pods based on a static desired number of text-to-speech batch requests per pod - requestsPerPod: 50 - # -- If you have custom metrics you would like to scale with, you may add them here. - # See the [k8s docs](https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/) - # for how to structure a list of metrics - custom: [] - - # -- [Configurable scaling behavior](https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#configurable-scaling-behavior) - # @default -- "*See values.yaml file for default*" - behavior: - scaleDown: - policies: - - type: Pods - value: 1 - periodSeconds: 60 - - type: Percent - value: 25 - periodSeconds: 60 - -api: - # -- namePrefix is the prefix to apply to the name of all K8s objects - # associated with the Deepgram API containers. - createContourGateway: true - ingress: - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: 'false' - nginx.ingress.kubernetes.io/ssl-redirect: 'false' - enabled: true - hosts: - - host: deepgram.int.meesho.int - paths: - - pathType: ImplementationSpecific - path: / - apiServiceName: deepgram-api-external - servicePort: 80 - ingressClassName: contour-internal-1 - servicePort: 80 - enableWebsocket: false - namePrefix: deepgram-api - namespace: dg-self-hosted-int - slowStart: - enabled: true - window: 60s - aggression: 0.5 - minPercent: 5 - - - image: - # -- path configures the image path to use for creating API containers. - # You may change this from the public Quay image path if you have imported - # Deepgram images into a private container registry. - path: quay.io/deepgram/self-hosted-api - # -- pullPolicy configures how the Kubelet attempts to pull the Deepgram API image - pullPolicy: IfNotPresent - # -- tag defines which Deepgram release to use for API containers - tag: release-260430 - - # -- Additional labels to add to API resources - additionalLabels: {} - - # -- (object) Additional annotations to add to the API deployment - additionalAnnotations: - - updateStrategy: - rollingUpdate: - # -- The maximum number of API pods, relative to the number of replicas, - # that can go offline during a rolling update. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-unavailable) - # for more details. - maxUnavailable: 0 - # -- The maximum number of extra API pods that can be created during a rollingUpdate, - # relative to the number of replicas. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-surge) - # for more details. - maxSurge: 1 - - # -- Configure resource limits per API container. See - # [Deepgram's documentation](https://developers.deepgram.com/docs/self-hosted-deployment-environments#api) - # for more details. - # @default -- `` - resources: - requests: - memory: "4Gi" - cpu: "2000m" - limits: - memory: "8Gi" - cpu: "4000m" - - # -- Readiness probe customization for API pods. - # @default -- `` - readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 1 - # -- Liveness probe customization for API pods. - # @default -- `` - livenessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 3 - - # -- [Affinity and anti-affinity](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#affinity-and-anti-affinity) - # to apply for API pods. - affinity: {} - # -- [Tolerations](https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/) - # to apply to API pods. - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepgram-api - - - # -- [Security context](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) for API pods. - securityContext: {} - - serviceAccount: - # -- Specifies whether to create a default service account for the Deepgram API Deployment. - create: true - # -- (string) Allows providing a custom service account name for the API component. - # If left empty, the default service account name will be used. - # If specified, and `api.serviceAccount.create = true`, this defines the name of the default service account. - # If specified, and `api.serviceAccount.create = false`, this provides the name of a preconfigured service account - # you wish to attach to the API deployment. - name: - - # -- Configure how the API will listen for your requests - # @default -- `` - server: - # baseUrl is the prefix requests to the API. - baseUrl: "/v1" - # -- host is the IP address to listen on. You will want to listen - # on all interfaces to interact with other pods in the cluster. - host: "0.0.0.0" - # -- port to listen on. - port: 8080 - - # -- callbackConnTimeout configures how long to wait for a connection to a callback URL. - # See [Deepgram's callback documentation](https://developers.deepgram.com/docs/callback) - # for more details. The value should be a humantime duration. - callbackConnTimeout: "1s" - # -- callbackTimeout configures how long to wait for a response from a callback URL. - # See [Deepgram's callback documentation](https://developers.deepgram.com/docs/callback) - # for more details. The value should be a humantime duration. - callbackTimeout: "10s" - - # -- fetchConnTimeout configures how long to wait for a connection to a fetch URL. - # The value should be a humantime duration. - # A fetch URL is a URL passed in an inference request from which a payload should be - # downloaded. - fetchConnTimeout: "1s" - # -- fetchTimeout configures how long to wait for a response from a fetch URL. - # The value should be a humantime duration. - # A fetch URL is a URL passed in an inference request from which a payload should be - # downloaded. - fetchTimeout: "60s" - - # -- Specify custom DNS resolution options. - # @default -- `` - resolver: - # -- nameservers allows for specifying custom domain name server(s). - # A valid list item's format is "{IP} {PORT} {PROTOCOL (tcp or udp)}", - # e.g. `"127.0.0.1 53 udp"`. - nameservers: [] - # -- (int) maxTTL sets the DNS TTL value if specifying a custom DNS nameserver. - maxTTL: - - # -- Enable ancillary features - # @default -- `` - features: - # -- Enables entity detection on pre-recorded audio - # *if* a valid entity detection model is available. - # *WARNING*: Beta functionality. - entityDetection: false - - # -- Enables entity-based redaction on pre-recorded audio - # *if* a valid entity detection model is available. - # *WARNING*: Beta functionality. - entityRedaction: false - - # -- If API is receiving requests faster than Engine can process them, a request - # queue will form. By default, this queue is stored in memory. Under high load, - # the queue may grow too large and cause Out-Of-Memory errors. To avoid this, - # set a diskBufferPath to buffer the overflow on the request queue to disk. - # - # WARN: This is only to temporarily buffer requests during high load. - # If there is not enough Engine capacity to process the queued requests over time, - # the queue (and response time) will grow indefinitely. - diskBufferPath: - - # -- driverPool configures the backend pool of speech engines (generically referred to as - # "drivers" here). The API will load-balance among drivers in the standard - # pool; if one standard driver fails, the next one will be tried. - # @default -- `` - driverPool: - # -- standard is the main driver pool to use. - # @default -- `` - standard: - # -- timeoutBackoff is the factor to increase the timeout by - # for each additional retry (for exponential backoff). - timeoutBackoff: 1.2 - - # -- retrySleep defines the initial sleep period (in humantime duration) - # before attempting a retry. - retrySleep: "2s" - # -- retryBackoff is the factor to increase the retrySleep - # by for each additional retry (for exponential backoff). - retryBackoff: 1.6 - - # -- Maximum response to deserialize from Driver (in bytes). - # Default is 1GB, expressed in bytes. - maxResponseSize: "1073741824" - -engine: - # -- namePrefix is the prefix to apply to the name of all K8s objects - # associated with the Deepgram Engine containers. - namePrefix: "deepgram-engine" - - image: - # -- path configures the image path to use for creating Engine containers. - # You may change this from the public Quay image path if you have imported - # Deepgram images into a private container registry. - path: quay.io/deepgram/self-hosted-engine - # -- pullPolicy configures how the Kubelet attempts to pull the Deepgram Engine image - pullPolicy: IfNotPresent - # -- tag defines which Deepgram release to use for Engine containers - tag: release-260430 - - # -- Additional labels to add to Engine resources - additionalLabels: {} - - # -- (object) Additional annotations to add to the Engine deployment - additionalAnnotations: - - updateStrategy: - rollingUpdate: - # -- The maximum number of Engine pods, relative to the number of replicas, - # that can go offline during a rolling update. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-unavailable) - # for more details. - maxUnavailable: 0 - # -- The maximum number of extra Engine pods that can be created during a rollingUpdate, - # relative to the number of replicas. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-surge) - # for more details. - maxSurge: 1 - - # -- Configure resource limits per Engine container. See - # [Deepgram's documentation](https://developers.deepgram.com/docs/self-hosted-deployment-environments#engine) - # for more details. - # @default -- `` - resources: - requests: - memory: "7" - cpu: "3" - # -- gpu maps to the nvidia.com/gpu resource parameter - gpu: 1 - limits: - memory: "8Gi" - cpu: "4" - # -- gpu maps to the nvidia.com/gpu resource parameter - gpu: 1 - - # -- The startupProbe combination of `periodSeconds` and `failureThreshold` allows - # time for the container to load all models and start listening for incoming requests. - # - # Model load time can be affected by hardware I/O speeds, as well as network speeds - # if you are using a network volume mount for the models. - # - # If you are hitting the failure threshold before models are finished loading, you may - # want to extend the startup probe. However, this will also extend the time it takes - # to detect a pod that can't establish a network connection to validate its license. - # @default -- `` - startupProbe: - # -- periodSeconds defines how often to execute the probe. - periodSeconds: 10 - # -- failureThreshold defines how many unsuccessful startup probe attempts - # are allowed before the container will be marked as Failed - failureThreshold: 60 - - # -- Readiness probe customization for Engine pods. - # @default -- `` - readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 1 - # -- Liveness probe customization for Engine pods. - # @default -- `` - livenessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 3 - - # -- [Affinity and anti-affinity](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#affinity-and-anti-affinity) - # to apply for Engine pods. - affinity: {} - # -- [Tolerations](https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/) - # to apply to Engine pods. - nodeSelector: - dedicated: deepgram-engine - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepgram-engine - - effect: NoSchedule - key: nvidia.com/gpu - operator: Equal - value: present - - effect: NoSchedule - key: nvidia.com/gpu - operator: Exists - - - # -- [Security context](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) for API pods. - securityContext: {} - - serviceAccount: - # -- Specifies whether to create a default service account for the Deepgram Engine Deployment. - create: true - # -- (string) Allows providing a custom service account name for the Engine component. - # If left empty, the default service account name will be used. - # If specified, and `engine.serviceAccount.create = true`, this defines the name of the default service account. - # If specified, and `engine.serviceAccount.create = false`, this provides the name of a preconfigured service account - # you wish to attach to the Engine deployment. - name: - - concurrencyLimit: - # -- (int) activeRequests limits the number of active requests handled by - # a single Engine container. - # If additional requests beyond the limit are sent, the API container - # forming the request will try a different Engine pod. If no Engine pods - # are able to accept the request, the API will return a 429 HTTP response - # to the client. The `nil` default means no limit will be set. - activeRequests: 12 - - # -- Configure Engine containers to listen for requests from API containers. - # @default -- `` - server: - # -- host is the IP address to listen on for inference requests. - # You will want to listen on all interfaces to interact with - # other pods in the cluster. - host: "0.0.0.0" - # -- port to listen on for inference requests - port: 8080 - - # -- metricsServer exposes an endpoint on each Engine container - # for reporting inference-specific system metrics. - # See https://developers.deepgram.com/docs/metrics-guide#deepgram-engine - # for more details. - # @default -- `` - metricsServer: - # -- host is the IP address to listen on for metrics requests. - # You will want to listen on all interfaces to interact with - # other pods in the cluster. - host: "0.0.0.0" - # -- port to listen on for metrics requests - port: 9273 - - modelManager: - volumes: - customVolumeClaim: - # -- You may manually create your own PersistentVolume and PersistentVolumeClaim to store and - # expose model files to the Deepgram Engine. Configure your storage beforehand, - # and enable here. - # Note: Make sure the PV and PVC accessMode are set to `readWriteMany` or `readOnlyMany` - enabled: false - # -- (string) Name of your pre-configured PersistentVolumeClaim - name: - # -- Name of the directory within your pre-configured PersistentVolume - # where the models are stored - modelsDirectory: "/" - - aws: - efs: - # -- Whether to use an [AWS Elastic File Sytem](https://aws.amazon.com/efs/) - # to store Deepgram models for use by Engine containers. - # This option requires your cluster to be running in - # [AWS EKS](https://aws.amazon.com/eks/). - enabled: false - # -- Name prefix for the resources associated with the model storage in AWS EFS. - namePrefix: dg-models - # -- (string) FileSystemId of existing AWS Elastic File System where - # Deepgram model files will be persisted. - # You can find it using the AWS CLI: - # ``` - # $ aws efs describe-file-systems --query "FileSystems[*].FileSystemId" - # ``` - fileSystemId: - # -- Whether to force a fresh download of all model links provided, - # even if models are already present in EFS. - forceDownload: false - nova3: - enabled: false - multilingual: - enabled: true - gcp: - gpd: - # -- Whether to use an [GKE Persistent Disks](https://cloud.google.com/kubernetes-engine/docs/concepts/persistent-volumes) - # to store Deepgram models for use by Engine containers. - # This option requires your cluster to be running in - # [GCP GKE](https://cloud.google.com/kubernetes-engine). - # See the GKE documentation on - # [using pre-existing persistent disks](https://cloud.google.com/kubernetes-engine/docs/how-to/persistent-volumes/preexisting-pd). - enabled: true - # -- Name prefix for the resources associated with the model storage in GCP GPD. - namePrefix: dg-models - # -- The storageClassName of the existing persistent disk. - storageClassName: "standard-rwo" - # -- The size of your pre-existing persistent disk. - storageCapacity: "40G" - # -- The identifier of your pre-existing persistent disk. - # The format is projects/{project_id}/zones/{zone_name}/disks/{disk_name} for Zonal persistent disks, - # or projects/{project_id}/regions/{region_name}/disks/{disk_name} for Regional persistent disks. - volumeHandle: "projects/meesho-shared-int-0525/zones/asia-southeast1-a/disks/deepgram-model-storage-nova3-multilingual" - fsType: "ext4" - - models: - # -- Links to your Deepgram models, if automatically downloading - # into storage backing a persistent volume. - # **Automatic downloads are currently supported for AWS EFS volumes only.** - # Insert each model link provided to you by your Deepgram - # Account Representative. - links: [] - - # -- chunking defines the size of audio chunks to process in seconds. - # Adjusting these values will affect both inference performance and accuracy - # of results. Please contact your Deepgram Account Representative if you - # want to adjust any of these values. - # @default -- `` - chunking: - speechToText: - batch: - # -- (float) minDuration is the minimum audio duration for a STT chunk size for a batch request - minDuration: - # -- (float) minDuration is the maximum audio duration for a STT chunk size for a batch request - maxDuration: - streaming: - # -- (float) minDuration is the minimum audio duration for a STT chunk size for a streaming request - minDuration: - # -- (float) minDuration is the maximum audio duration for a STT chunk size for a streaming request - maxDuration: - # -- step defines how often to return interim results, in seconds. - # This value may be lowered to increase the frequency of interim results. - # However, this also causes a significant decrease in the number of concurrent - # streams supported by a single GPU. Please contact your Deepgram Account - # representative for more details. - step: 0.2 - - halfPrecision: - # -- Engine will automatically enable half precision operations if your GPU supports - # them. You can explicitly enable or disable this behavior with the state parameter - # which supports `"enable"`, `"disabled"`, and `"auto"`. - state: "auto" - -# -- Configuration options for the optional -# [Deepgram License Proxy](https://developers.deepgram.com/docs/license-proxy). -# @default -- `` -licenseProxy: - # -- The License Proxy is optional, but highly recommended to be deployed in production - # to enable highly available environments. - enabled: false - - # -- If the License Proxy is deployed, one replica should be sufficient to - # support many API/Engine pods. - # Highly available environments may wish to deploy a second replica to ensure - # uptime, which can be toggled with this option. - deploySecondReplica: false - - # -- Even with a License Proxy deployed, API and Engine pods can be configured to keep the - # upstream `license.deepgram.com` license server as a fallback licensing option if the - # License Proxy is unavailable. - # Disable this option if you are restricting API/Engine Pod network access for security reasons, - # and only the License Proxy should send egress traffic to the upstream license server. - keepUpstreamServerAsBackup: true - - # -- namePrefix is the prefix to apply to the name of all K8s objects - # associated with the Deepgram License Proxy containers. - namePrefix: "deepgram-license-proxy" - - image: - # -- path configures the image path to use for creating License Proxy containers. - # You may change this from the public Quay image path if you have imported - # Deepgram images into a private container registry. - path: quay.io/deepgram/self-hosted-license-proxy - # -- tag defines which Deepgram release to use for License Proxy containers - tag: release-260430 - # -- pullPolicy configures how the Kubelet attempts to pull the Deepgram - # License Proxy image - pullPolicy: IfNotPresent - - # -- Additional labels to add to License Proxy resources - additionalLabels: {} - - # -- (object) Additional annotations to add to the LicenseProxy deployment - additionalAnnotations: - - updateStrategy: - # -- For the LicenseProxy, we only expose maxSurge and not maxUnavailable. - # This is to avoid accidentally having all LicenseProxy nodes go offline during upgrades, - # which could impact the entire cluster's connection to the Deepgram License Server. - # @default -- `` - rollingUpdate: - # -- The maximum number of extra License Proxy pods that can be created during a rollingUpdate, - # relative to the number of replicas. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-surge) - # for more details. - maxSurge: 1 - - # -- Configure resource limits per License Proxy container. See - # [Deepgram's documentation](https://developers.deepgram.com/docs/license-proxy#system-requirements) - # for more details. - # @default -- `` - resources: - requests: - memory: "1Gi" - cpu: "1000m" - limits: - memory: "8Gi" - cpu: "2000m" - - # -- Readiness probe customization for License Proxy pods. - # @default -- `` - readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 1 - # -- Liveness probe customization for Proxy pods. - # @default -- `` - livenessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 3 - - # -- [Affinity and anti-affinity](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#affinity-and-anti-affinity) - # to apply for License Proxy pods. - affinity: {} - # -- [Tolerations](https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/) - # to apply to License Proxy pods. - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepgram-proxy-pool - - # -- [Security context](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) for API pods. - securityContext: {} - - serviceAccount: - # -- Specifies whether to create a default service account for the Deepgram License Proxy Deployment. - create: true - # -- (string) Allows providing a custom service account name for the LicenseProxy component. - # If left empty, the default service account name will be used. - # If specified, and `licenseProxy.serviceAccount.create = true`, this defines the name of the default service account. - # If specified, and `licenseProxy.serviceAccount.create = false`, this provides the name of a preconfigured service account - # you wish to attach to the License Proxy deployment. - name: - - # -- Configure how the license proxy will listen for licensing requests. - # @default -- `` - server: - # --host is the IP address to listen on. You will want to listen - # on all interfaces to interact with other pods in the cluster. - host: "0.0.0.0" - # -- port to listen on. - port: 8443 - - # -- baseUrl is the prefix for incoming license verification requests. - baseUrl: "/" - - # -- statusPort is the port to listen on for the status/health endpoint. - statusPort: 8080 - -# -- Passthrough values for [NVIDIA GPU Operator Helm chart](https://github.com/NVIDIA/gpu-operator/blob/master/deployments/gpu-operator/values.yaml) -# You may use the NVIDIA GPU Operator to manage installation of NVIDIA drivers and the container toolkit on nodes with attached GPUs. -# @default -- `` -gpu-operator: - # -- Whether to install the NVIDIA GPU Operator to manage driver and/or container toolkit installation. - # See the list of [supported Operating Systems](https://docs.nvidia.com/datacenter/cloud-native/gpu-operator/latest/platform-support.html#supported-operating-systems-and-kubernetes-platforms) - # to verify compatibility with your cluster/nodes. Disable this option if your cluster/nodes are not compatible. - # If disabled, you will need to self-manage NVIDIA software installation on all nodes where you want - # to schedule Deepgram Engine pods. - enabled: false - driver: - # -- Whether to install NVIDIA drivers on nodes where a NVIDIA GPU is detected. - # If your Kubernetes nodes run a base image that comes with NVIDIA drivers pre-configured, - # disable this option, but keep the parent `gpu-operator` and sibling `toolkit` - # options enabled. - enabled: true - # -- NVIDIA driver version to install. - version: "550.54.15" - toolkit: - # -- Whether to install NVIDIA drivers on nodes where a NVIDIA GPU is detected. - enabled: true - # -- NVIDIA container toolkit to install. The default `ubuntu` image tag for the - # toolkit requires a dynamic runtime link to a version of GLIBC that may not be - # present on nodes running older Linux distribution releases, such as Ubuntu 22.04. - # Therefore, we specify the `ubi8` image, which statically links the GLIBC library - # and avoids this issue. - version: v1.15.0-ubi8 - -cluster-autoscaler: - # -- Set to `true` to enable node autoscaling with AWS EKS. Note needed for GKE, as autoscaling is enabled by a - # [cli option on cluster creation](https://cloud.google.com/kubernetes-engine/docs/how-to/cluster-autoscaler#creating_a_cluster_with_autoscaling). - enabled: false - rbac: - serviceAccount: - # -- Name of the IAM Service Account with the [necessary permissions](https://github.com/kubernetes/autoscaler/blob/master/cluster-autoscaler/cloudprovider/aws/README.md#permissions) - name: cluster-autoscaler-sa - annotations: - # -- (string) Replace with the AWS Role ARN configured for the Cluster Autoscaler. - # See the [Deepgram AWS EKS guide](https://developers.deepgram.com/docs/aws-k8s#creating-a-cluster) - # or [Cluster Autoscaler AWS documentation](https://github.com/kubernetes/autoscaler/blob/master/cluster-autoscaler/cloudprovider/aws/README.md#permissions) - # for details. - eks.amazonaws.com/role-arn: - autoDiscovery: - # -- (string) Name of your AWS EKS cluster. Using the [Cluster Autoscaler](https://github.com/kubernetes/autoscaler) - # on AWS requires knowledge of certain cluster metadata. - clusterName: - # -- (string) Region of your AWS EKS cluster. Using the [Cluster Autoscaler](https://github.com/kubernetes/autoscaler) - # on AWS requires knowledge of certain cluster metadata. - awsRegion: - -# -- Passthrough values for [Prometheus k8s stack Helm chart](https://github.com/prometheus-community/helm-charts/tree/main/charts/kube-prometheus-stack). -# Prometheus (and its adapter) should be configured when scaling.auto is enabled. -# You may choose to use the installation/configuration bundled in this Helm chart, -# or you may configure an existing Prometheus installation in your cluster to expose -# the needed values. -# See source Helm chart for explanation of available values. Default values provided in this chart are used -# to provide pod autoscaling for Deepgram pods. -# @default -- `` -kube-prometheus-stack: - # -- (bool) Normally, this chart will be installed if `scaling.auto.enabled` is true. However, if you wish - # to manage the Prometheus adapter in your cluster on your own and not as part of the Deepgram Helm chart, - # you can force it to not be installed by setting this to `false`. - includeDependency: false - - fullnameOverride: "dg-prometheus-stack" - prometheus: - prometheusSpec: - additionalScrapeConfigs: - - job_name: "dg_engine_metrics" - scrape_interval: "2s" - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - "{{ .Release.Namespace }}" - relabel_configs: - - source_labels: [__meta_kubernetes_service_name] - regex: "(.*)-metrics" - action: keep - - source_labels: [__meta_kubernetes_endpoint_port_name] - regex: "metrics" - action: keep - - source_labels: [__meta_kubernetes_namespace] - target_label: namespace - - source_labels: [__meta_kubernetes_service_name] - target_label: service - - source_labels: [__meta_kubernetes_pod_name] - target_label: pod - - prometheusOperator: - enabled: false - - alertmanager: - enabled: false - - grafana: - enabled: false - - nodeExporter: - enabled: false - - kube-state-metrics: - enabled: false - metricLabelsAllowlist: - - namespaces=[{{ .Release.Namespace }}],deployments=[app] - -# -- Passthrough values for [Prometheus Adapter Helm chart](https://github.com/prometheus-community/helm-charts/tree/main/charts/prometheus-adapter). -# Prometheus, and its adapter here, should be configured when scaling.auto is enabled. -# You may choose to use the installation/configuration bundled in this Helm chart, -# or you may configure an existing Prometheus installation in your cluster to expose -# the needed values. -# See source Helm chart for explanation of available values. Default values provided in this chart are used -# to provide pod autoscaling for Deepgram pods. -# @default -- `` -prometheus-adapter: - # -- Normally, this chart will be installed if `scaling.auto.enabled` is true. However, if you wish - # to manage the Prometheus adapter in your cluster on your own and not as part of the Deepgram Helm chart, - # you can force it to not be installed by setting this to `false`. - includeDependency: - prometheus: - url: http://dg-prometheus-stack-prometheus.{{ .Release.Namespace }}.svc - rules: - default: false - external: - - name: - as: "engine_active_requests_stt_streaming" - seriesQuery: 'engine_active_requests{kind="stream"}' - metricsQuery: 'avg(engine_active_requests{kind="stream"})' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_active_requests_stt_batch" - seriesQuery: 'engine_active_requests{kind="batch"}' - metricsQuery: 'avg(engine_active_requests{kind="batch"})' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_active_requests_tts_batch" - seriesQuery: 'engine_active_requests{kind="tts"}' - metricsQuery: 'avg(engine_active_requests{kind="tts"})' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_estimated_stream_capacity" - seriesQuery: 'engine_active_requests{kind="stream"}' - metricsQuery: 'avg_over_time((sum(engine_active_requests{kind="stream"}) / sum(engine_estimated_stream_capacity) * 100)[1m:1m])' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_requests_active_to_max_ratio" - seriesQuery: "engine_max_active_requests" - metricsQuery: "avg_over_time((sum(engine_active_requests) / sum(engine_max_active_requests) * 100)[1m:1m])" - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_to_api_pod_ratio" - seriesQuery: 'kube_deployment_labels{label_app="deepgram-engine"}' - metricsQuery: '(sum(kube_deployment_status_replicas and on(deployment) kube_deployment_labels{label_app="deepgram-engine"})) / (sum(kube_deployment_status_replicas and on(deployment) kube_deployment_labels{label_app="deepgram-api"}))' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } diff --git a/helm-overrides/k8s-shared-int-ase1/external-secrets/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/external-secrets/custom-values.yaml deleted file mode 100644 index f724c82..0000000 --- a/helm-overrides/k8s-shared-int-ase1/external-secrets/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Additional service annotations - annotations: {} - - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: cc-shared-devops-od - nodeSelector: - cloud.google.com/compute-class: cc-shared-devops-od - certController: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: cc-shared-devops-od - nodeSelector: - cloud.google.com/compute-class: cc-shared-devops-od - webhook: - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: cc-shared-devops-od - nodeSelector: - cloud.google.com/compute-class: cc-shared-devops-od \ No newline at end of file diff --git a/helm-overrides/k8s-shared-int-ase1/fluentd/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/fluentd/custom-values.yaml deleted file mode 100644 index 66ed4f3..0000000 --- a/helm-overrides/k8s-shared-int-ase1/fluentd/custom-values.yaml +++ /dev/null @@ -1,835 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-shrd-srsre-fluentd-int@meesho-shared-int-0525.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: shared - team: sre - type: fluentd - service: fluentd-shared-int - priority: p0 - env: int - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: shared - team: sre - type: fluentd - service: fluentd-shared-int - priority: p0 - env: int - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: [] -# - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - from_encoding "utf-8" - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @DG_SELF_HOSTED - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-shared-int-ase1/ingress-nginx-internal/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/ingress-nginx-internal/custom-values.yaml deleted file mode 100644 index 16667ae..0000000 --- a/helm-overrides/k8s-shared-int-ase1/ingress-nginx-internal/custom-values.yaml +++ /dev/null @@ -1,42 +0,0 @@ -ingress-nginx: - controller: - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - resources: - requests: - cpu: 100m - memory: 768Mi - autoscaling: - enabled: true - minReplicas: 1 - maxReplicas: 15 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 70 - extraArgs: - enable-ssl-passthrough: true - tcp-services-configmap: nginx-shared-internal/tcp-services - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-shared-internal-int"}}}' - nodeSelector: - cloud.google.com/compute-class: cc-contour-vmagent-od - tolerations: - - key: cloud.google.com/compute-class - operator: "Equal" - value: cc-contour-vmagent-od - effect: "NoSchedule" - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a \ No newline at end of file diff --git a/helm-overrides/k8s-shared-int-ase1/keda/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/keda/custom-values.yaml deleted file mode 100644 index 28fb871..0000000 --- a/helm-overrides/k8s-shared-int-ase1/keda/custom-values.yaml +++ /dev/null @@ -1,98 +0,0 @@ -keda: - operator: - replicaCount: 2 - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a - metricsServer: - replicaCount: 2 - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a - webhook: - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a - nodeSelector: - cloud.google.com/compute-class: cc-shared-devops-od - tolerations: - - key: cloud.google.com/compute-class - operator: "Equal" - value: "cc-shared-devops-od" - effect: "NoSchedule" - podLabels: - bu: "shared" - team: "shared-devops" - metricsAdapter: - bu: "shared" - team: "shared-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 1600Mi - requests: - cpu: 250m - memory: 850Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 60m - memory: 200Mi - metricServer: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 100m - memory: 200Mi - podAnnotations: - # -- Pod annotations for KEDA operator - keda: - prometheus.io/path: /metrics - prometheus.io/port: '8080' - prometheus.io/scrape: 'true' - # -- Pod annotations for KEDA Metrics Adapter - metricsAdapter: - prometheus.io/path: /metrics - prometheus.io/port: '8080' - prometheus.io/scrape: 'true' - # -- Pod annotations for KEDA Admission webhooks - webhooks: - prometheus.io/path: /metrics - prometheus.io/port: '8080' - prometheus.io/scrape: 'true' - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' diff --git a/helm-overrides/k8s-shared-int-ase1/kube-dns/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/kube-dns/custom-values.yaml deleted file mode 100644 index 7170324..0000000 --- a/helm-overrides/k8s-shared-int-ase1/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.225.0.2"],"prd.mrouter.int.svc.cluster.local":["10.225.0.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.225.0.2"]} diff --git a/helm-overrides/k8s-shared-int-ase1/kube-events/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/kube-events/custom-values.yaml deleted file mode 100644 index dae1fb8..0000000 --- a/helm-overrides/k8s-shared-int-ase1/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-shared-int - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - cloud.google.com/compute-class: cc-shared-devops-od - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: cc-shared-devops-od - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: cc-shared-devops-od - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: cc-shared-devops-od - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - cloud.google.com/compute-class: cc-shared-devops-od - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - value: cc-shared-devops-od - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-shared-int-ase1/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 4051712..0000000 --- a/helm-overrides/k8s-shared-int-ase1/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,486 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-shared-int - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-shared-int.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "shared" - team: "shared-sre" - service: "kube-state-metrics-shared-int" - env: "int" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - cloud.google.com/compute-class: "cc-shared-devops-od" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: cloud.google.com/compute-class - operator: "Equal" - value: "cc-shared-devops-od" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 20m - memory: 1.5Gi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-shared-int-ase1/kubectl-mcp-server/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/kubectl-mcp-server/custom-values.yaml deleted file mode 100644 index f3dd84e..0000000 --- a/helm-overrides/k8s-shared-int-ase1/kubectl-mcp-server/custom-values.yaml +++ /dev/null @@ -1,120 +0,0 @@ -fullnameOverride: "kubectl-mcp-server" - -replicas: 1 - -image: - # In-house hardened build (helm-templates/kubectl-mcp-server/docker/) — - # NOT the upstream Docker Hub image. 266→18 HIGH/CRIT vulns, non-root, - # multi-stage slim base, kubectl v1.33.12 / helm v3.21.0. - # digest: sha256:1b477e6eb78031e3da69e1536b55c966e184401e66b4e43471f301abb2016032 - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/prd/devop/kubectl-mcp-server - tag: "v2" - pullPolicy: IfNotPresent - -labels: - bu: infra - team: devops - service: kubectl-mcp-server - env: int - -serviceAccount: - create: true - annotations: {} - -rbac: - create: true - -podAnnotations: {} - -podSecurityContext: {} - -securityContext: - runAsNonRoot: true - runAsUser: 1000 - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -priorityClassName: "" - -# k8s-shared-int-ase1 is a shared NAP/compute-class cluster (no dedicated -# -devops node pools) — schedule on the shared devops compute class. -nodeSelector: - cloud.google.com/compute-class: cc-shared-devops-od - -tolerations: - - key: "cloud.google.com/compute-class" - operator: "Equal" - value: "cc-shared-devops-od" - effect: NoSchedule - -affinity: {} - -resources: - requests: - cpu: 250m - memory: 512Mi - limits: - cpu: 1000m - memory: 1Gi - -mcp: - mode: single - transport: http - host: "0.0.0.0" - port: 8000 - -auth: - allowAnonymous: false - -externalSecrets: - enabled: true - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/int/cntr/devop/kubectl-mcp-server" - -# tcpSocket on purpose — streamable-http transport exposes only /mcp, -# there is no /health route (see chart values.yaml note). -livenessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - -readinessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - -service: - port: 8000 - type: ClusterIP - -# Contour standard (2.0.0-style): renders HTTPProxy parent/child + -intra -# instead of a networking.k8s.io Ingress. -createContourGateway: true - -ingress: - enabled: true - ingressClassName: contour-internal-1 - servicePortNumber: 8000 - hosts: - - host: kubectl-mcp-server.int.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 diff --git a/helm-overrides/k8s-shared-int-ase1/kyverno/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/kyverno/custom-values.yaml deleted file mode 100644 index eb1c11e..0000000 --- a/helm-overrides/k8s-shared-int-ase1/kyverno/custom-values.yaml +++ /dev/null @@ -1,2240 +0,0 @@ -global: - - # -- Internal settings used with `helm template` to generate install manifest - # @ignored - templating: - enabled: false - debug: false - version: ~ - - image: - # -- (string) Global value that allows to set a single image registry across all deployments. - # When set, it will override any values set under `.image.registry` across the chart. - registry: ~ - # -- (list) Global list of Image pull secrets - # When set, it will override any values set under `imagePullSecrets` under different components across the chart. - imagePullSecrets: [] - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - caCertificates: - # -- Global CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - # Individual controller values will override this global value - data: ~ - - # -- Global value to set single volume to be mounted for CA certificates for all deployments. - # Not used when `.Values.global.caCertificates.data` is defined - # Individual controller values will override this global value - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Additional container environment variables to apply to all containers and init containers - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Global node labels for pod assignment. Non-global values will override the global value. - nodeSelector: - dedicated: shared-kyverno - - # -- Global List of node taints to tolerate. Non-global values will override the global value. - tolerations: - - key: "dedicated" - operator: "Equal" - value: "shared-kyverno" - effect: "NoSchedule" - -# -- (string) Override the name of the chart -nameOverride: ~ - -# -- (string) Override the expanded name of the chart -fullnameOverride: ~ - -# -- (string) Override the namespace the chart deploys to -namespaceOverride: ~ - -upgrade: - # -- Upgrading from v2 to v3 is not allowed by default, set this to true once changes have been reviewed. - fromV2: false - -apiVersionOverride: - # -- (string) Override api version used to create `PodDisruptionBudget`` resources. - # When not specified the chart will check if `policy/v1/PodDisruptionBudget` is available to - # determine the api version automatically. - podDisruptionBudget: ~ - -rbac: - roles: - # -- Aggregate ClusterRoles to Kubernetes default user-facing roles. For more information, see [User-facing roles](https://kubernetes.io/docs/reference/access-authn-authz/rbac/#user-facing-roles) - aggregate: - admin: true - view: true - -# Use openreports.io as the API group for reporting -openreports: - # -- Enable OpenReports feature in controllers - enabled: false - # -- Whether to install CRDs from the upstream OpenReports chart. Setting this to true requires enabled to also be true. - installCrds: false - -# CRDs configuration -crds: - - # -- Whether to have Helm install the Kyverno CRDs, if the CRDs are not installed by Helm, they must be added before policies can be created - install: true - - reportsServer: - # -- Kyverno reports-server is used in your cluster - enabled: false - - groups: - - # -- Install CRDs in group `kyverno.io` - kyverno: - cleanuppolicies: true - clustercleanuppolicies: true - clusterpolicies: true - globalcontextentries: true - policies: true - policyexceptions: true - updaterequests: true - - # -- Install CRDs in group `policies.kyverno.io` - policies: - validatingpolicies: true - policyexceptions: true - imagevalidatingpolicies: true - namespacedimagevalidatingpolicies: true - mutatingpolicies: true - generatingpolicies: true - deletingpolicies: true - namespaceddeletingpolicies: true - namespacedvalidatingpolicies: true - - # -- Install CRDs in group `reports.kyverno.io` - reports: - clusterephemeralreports: true - ephemeralreports: true - - # -- Install CRDs in group `wgpolicyk8s.io` - wgpolicyk8s: - clusterpolicyreports: false ##lk - policyreports: false ##lk - - # -- Additional CRDs annotations (Replace=true avoids 262144-byte annotation limit on large CRDs) - annotations: - argocd.argoproj.io/sync-options: Replace=true - # strategy.spinnaker.io/replace: 'true' - - # -- Additional CRDs labels - customLabels: {} - - migration: - - # -- Enable CRDs migration using helm post upgrade hook - enabled: true - - # -- Resources to migrate - resources: - - cleanuppolicies.kyverno.io - - clustercleanuppolicies.kyverno.io - - clusterpolicies.kyverno.io - - globalcontextentries.kyverno.io - - policies.kyverno.io - - policyexceptions.kyverno.io - - updaterequests.kyverno.io - - deletingpolicies.policies.kyverno.io - - generatingpolicies.policies.kyverno.io - - imagevalidatingpolicies.policies.kyverno.io - - namespacedimagevalidatingpolicies.policies.kyverno.io - - mutatingpolicies.policies.kyverno.io - - namespaceddeletingpolicies.policies.kyverno.io - - namespacedvalidatingpolicies.policies.kyverno.io - - policyexceptions.policies.kyverno.io - - validatingpolicies.policies.kyverno.io - - image: - # -- (string) Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- (string) Image repository - repository: kyverno/kyverno-cli - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- (string) Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podResources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -# Configuration -config: - - # -- Create the configmap. - create: true - - # -- Preserve the configmap settings during upgrade. - preserve: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - # -- Enable registry mutation for container images. Enabled by default. - enableDefaultRegistryMutation: true - - # -- The registry hostname used for the image mutation. - defaultRegistry: docker.io - - # -- Exclude groups - excludeGroups: - - system:nodes - - # -- Exclude usernames - excludeUsernames: [] - # - '!system:kube-scheduler' - - # -- Exclude roles - excludeRoles: [] - - # -- Exclude roles - excludeClusterRoles: [] - - # -- Generate success events. - generateSuccessEvents: false - - # -- Resource types to be skipped by the Kyverno policy engine. - # Make sure to surround each entry in quotes so that it doesn't get parsed as a nested YAML list. - # These are joined together without spaces, run through `tpl`, and the result is set in the config map. - # @default -- See [values.yaml](values.yaml) - resourceFilters: - - '[Event,*,*]' - # - '[*/*,kube-system,*]' - - '[*/*,kube-public,*]' - - '[*/*,kube-node-lease,*]' - - '[Node,*,*]' - - '[Node/?*,*,*]' - - '[APIService,*,*]' - - '[APIService/?*,*,*]' - - '[TokenReview,*,*]' - - '[SubjectAccessReview,*,*]' - - '[SelfSubjectAccessReview,*,*]' - - '[Binding,*,*]' - - '[Pod/binding,*,*]' - - '[ReplicaSet,*,*]' - - '[ReplicaSet/?*,*,*]' - - '[EphemeralReport,*,*]' - - '[ClusterEphemeralReport,*,*]' - # exclude resources from the chart - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:additional]' - - '[ClusterRoleBinding,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.configMapName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.metricsConfigMapName" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Job,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[Job/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.admission-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.background-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - # -- Sets the threshold for the total number of UpdateRequests generated for mutateExisitng and generate policies. - updateRequestThreshold: 1000 - - # -- Defines the `namespaceSelector`/`objectSelector` in the webhook configurations. - # The Kyverno namespace is excluded if `excludeKyvernoNamespace` is `true` (default) - webhooks: - # Exclude namespaces - # namespaceSelector: lk-s2 - # matchExpressions: - # - key: kubernetes.io/metadata.name - # operator: NotIn - # values: - # - kube-systemm lk-s2 - # Exclude objects - # objectSelector: - # matchExpressions: - # - key: webhooks.kyverno.io/exclude - # operator: DoesNotExist - - # -- Defines annotations to set on webhook configurations. - webhookAnnotations: - # Example to disable admission enforcer on AKS: - # 'admissions.enforcer/disabled': 'true' lk-s3 - - # -- Defines labels to set on webhook configurations. - webhookLabels: {} - # Example to adopt webhook resources in ArgoCD: - # 'argocd.argoproj.io/instance': 'kyverno' - - # -- Defines match conditions to set on webhook configurations (requires Kubernetes 1.27+). - matchConditions: [] - - # -- Exclude Kyverno namespace - # Determines if default Kyverno namespace exclusion is enabled for webhooks and resourceFilters - excludeKyvernoNamespace: true - - # -- resourceFilter namespace exclude - # Namespaces to exclude from the default resourceFilters - resourceFiltersExcludeNamespaces: [] - - # -- resourceFilters exclude list - # Items to exclude from config.resourceFilters - resourceFiltersExclude: [] - - # -- resourceFilter namespace include - # Namespaces to include to the default resourceFilters - resourceFiltersIncludeNamespaces: [] - - # -- resourceFilters include list - # Items to include to config.resourceFilters - resourceFiltersInclude: [] - -# Metrics configuration -metricsConfig: - - # -- Create the configmap. - create: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - namespaces: - - # -- List of namespaces to capture metrics for. - include: [] - - # -- list of namespaces to NOT capture metrics for. - exclude: [] - - # -- (string) Rate at which metrics should reset so as to clean up the memory footprint of kyverno metrics, if you might be expecting high memory footprint of Kyverno's metrics. Default: 0, no refresh of metrics. WARNING: This flag is not working since Kyverno 1.8.0 - metricsRefreshInterval: ~ - # metricsRefreshInterval: 24h - - # -- (list) Configures the bucket boundaries for all Histogram metrics, changing this configuration requires restart of the kyverno admission controller - bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 15, 20, 25, 30] - - # -- (map) Configures the exposure of individual metrics, by default all metrics and all labels are exported, changing this configuration requires restart of the kyverno admission controller - metricsExposure: - kyverno_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_image_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_mutating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_generating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_admission_review_duration_seconds: - # enabled: false - disabledLabelDimensions: ["resource_namespace"] - kyverno_policy_rule_info_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_policy_results_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_admission_requests_total: - disabledLabelDimensions: ["resource_namespace"] - kyverno_cleanup_controller_deletedobjects_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - -# -- Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -imagePullSecrets: {} - # regcred: - # registry: foo.example.com - # username: foobar - # password: secret - # regcred2: - # registry: bar.example.com - # username: barbaz - # password: secret2 - -# -- Existing Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -existingImagePullSecrets: [] - # - test-registry - # - other-test-registry - -# Tests configuration -test: - # -- Sleep time before running test - sleep: 20 - - image: - # -- (string) Image registry - registry: curlimages - # -- Image repository - repository: curl - # -- Image tag - # Defaults to `latest` if omitted - tag: '8.10.1' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - # -- Security context for the test containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- Additional Pod annotations - podAnnotations: {} - - # -- List of node taints to tolerate - tolerations: [] - -# -- Additional labels -customLabels: {} - - -webhooksCleanup: - # -- Create a helm pre-delete hook to cleanup webhooks. - enabled: true - - autoDeleteWebhooks: - # -- Allow webhooks controller to delete webhooks using finalizers - enabled: false - - image: - # -- (string) Image registry - registry: registry.k8s.io - # -- Image repository - repository: kubectl - # -- Image tag - # Defaults to `latest` if omitted - tag: 'v1.32.7' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -grafana: - # -- Enable grafana dashboard creation. - enabled: false - - # -- Configmap name template. - configMapName: '{{ include "kyverno.fullname" . }}-grafana' - - # -- (string) Namespace to create the grafana dashboard configmap. - # If not set, it will be created in the same namespace where the chart is deployed. - namespace: ~ - - # -- Grafana dashboard configmap annotations. - annotations: {} - - # -- Grafana dashboard configmap labels - labels: - grafana_dashboard: "1" - - # -- create GrafanaDashboard custom resource referencing to the configMap. - # according to https://grafana-operator.github.io/grafana-operator/docs/examples/dashboard_from_configmap/readme/ - grafanaDashboard: - create: false - folder: kyverno - allowCrossNamespaceImport: true - matchLabels: - dashboards: "grafana" - -# Features configuration -features: - admissionReports: - # -- Enables the feature - enabled: true - aggregateReports: - # -- Enables the feature - enabled: true - policyReports: - # -- Enables the feature - enabled: false ##lk - validatingAdmissionPolicyReports: - # -- Enables the feature - enabled: true - mutatingAdmissionPolicyReports: - # -- Enables the feature - enabled: false - reporting: - # -- Enables the feature - validate: true - # -- Enables the feature - mutate: true - # -- Enables the feature - mutateExisting: true - # -- Enables the feature - imageVerify: true - # -- Enables the feature - generate: true - autoUpdateWebhooks: - # -- Enables the feature - enabled: true - backgroundScan: - # -- Enables the feature - enabled: true - # -- Number of background scan workers - backgroundScanWorkers: 2 - # -- Background scan interval - backgroundScanInterval: 1h - # -- Skips resource filters in background scan - skipResourceFilters: true - configMapCaching: - # -- Enables the feature - enabled: true - controllerRuntimeMetrics: - # -- Bind address for controller-runtime metrics (use "0" to disable it) - bindAddress: ":8080" - deferredLoading: - # -- Enables the feature - enabled: true - dumpPayload: - # -- Enables the feature - enabled: false - forceFailurePolicyIgnore: - # -- Enables the feature - enabled: false - generateValidatingAdmissionPolicy: - # -- Enables the feature - enabled: true - generateMutatingAdmissionPolicy: - # -- Enables the feature - enabled: false - dumpPatches: - # -- Enables the feature - enabled: false - globalContext: - # -- Maximum allowed response size from API Calls. A value of 0 bypasses checks (not recommended) - maxApiCallResponseLength: 2000000 - logging: - # -- Logging format - format: text - # -- Logging verbosity - verbosity: 2 - omitEvents: - # -- Events which should not be emitted (possible values `PolicyViolation`, `PolicyApplied`, `PolicyError`, and `PolicySkipped`) - eventTypes: - - PolicyApplied - - PolicySkipped - # - PolicyViolation - # - PolicyError - policyExceptions: - # -- Enables the feature - enabled: true - # -- Restrict policy exceptions to a single namespace - # Set to "*" to allow exceptions in all namespaces - namespace: '' - protectManagedResources: - # -- Enables the feature - enabled: false - registryClient: - # -- Allow insecure registry - allowInsecure: false - # -- Enable registry client helpers - credentialHelpers: - - default - - google - - amazon - - azure - - github - ttlController: - # -- Reconciliation interval for the label based cleanup manager - reconciliationInterval: 1m - tuf: - # -- Enables the feature - enabled: false - # -- (string) Path to Tuf root - root: ~ - # -- (string) Raw Tuf root - rootRaw: ~ - # -- (string) Tuf mirror - mirror: ~ - -# Admission controller configuration -admissionController: - autoscaling: - # -- Enable horizontal pod autoscaling - enabled: true - - # -- Minimum number of pods - minReplicas: 1 - - # -- Maximum number of pods - maxReplicas: 10 - - # -- Target CPU utilization percentage - targetCPUUtilizationPercentage: 80 - - # -- Configurable scaling behavior - behavior: {} - - # -- Overrides features defined at the root level - featuresOverride: - admissionReports: - # -- Max number of admission reports allowed in flight until the admission controller stops creating new ones - backPressureThreshold: 1000 - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- The ServiceAccount name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno admission controller activities. - # This will help ensure Kyverno stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- admissionController webhook server port - # in case you are using hostNetwork: true, you might want to change the port the webhookServer is listening to - webhookServer: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - admission-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.admissionController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - initContainer: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyvernopre - # -- (string) Image tag - # If missing, defaults to image.tag - tag: ~ - # -- (string) Image pull policy - # If missing, defaults to image.pullPolicy - pullPolicy: ~ - - resources: - # -- Pod resource limits - # limits: - # cpu: 100m - # memory: 256Mi - # -- Pod resource requests - requests: - cpu: 1000m - memory: 1Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - container: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyverno - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - resources: - # -- Pod resource limits - # limits: - # memory: 384Mi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Array of extra init containers - extraInitContainers: [] - # - name: init-container - # image: busybox - # command: ['sh', '-c', 'echo Hello'] - - # -- Array of extra containers to run alongside kyverno - extraContainers: [] - # - name: myapp-container - # image: busybox - # command: ['sh', '-c', 'echo Hello && sleep 3600'] - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Kyverno's metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Background controller configuration -backgroundController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable background controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: - - apiGroups: - - networking.k8s.io - resources: - - ingresses - - ingressclasses - - networkpolicies - verbs: - - create - - update - - patch - - delete - - apiGroups: - - rbac.authorization.k8s.io - resources: - - rolebindings - - roles - verbs: - - create - - update - - patch - - delete - - apiGroups: - - '' - resources: - - configmaps - - resourcequotas - - limitranges - verbs: - - create - - update - - patch - - delete - - apiGroups: - - resource.k8s.io - resources: - - resourceclaims - - resourceclaimtemplates - verbs: - - create - - delete - - update - - patch - - deletecollection - - apiGroups: - - autoscaling - resources: - - horizontalpodautoscalers - verbs: - - get - - list - - watch - - update - - patch - - apiGroups: - - apps - resources: - - deployments - - statefulsets - - replicasets - verbs: - - get - - list - - watch - - update - - patch - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - # - patch - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/background-controller - # -- Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - # limits: - # memory: 128Mi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - background-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.backgroundController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - # -- backgroundController server port - # in case you are using hostNetwork: true, you might want to change the port the backgroundController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Cleanup controller configuration -cleanupController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable cleanup controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - delete - # - list - # - watch - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/cleanup-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- cleanupController server port - # in case you are using hostNetwork: true, you might want to change the port the cleanupController is listening to - server: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - # limits: - # memory: 128Mi - # -- Pod resource requests - requests: - cpu: 1100m - memory: 1.2Gi - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - cleanup-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `service.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Reports controller configuration -reportsController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable reports controller. - enabled: false ##lk - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/reports-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno reports controller activities. - # This will help ensure Kyverno reports stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - memory: 128Mi - # -- Pod resource requests - requests: - cpu: 100m - memory: 64Mi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - reports-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.reportsController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- (string) Service node port. - # Only used if `type` is `NodePort`. - nodePort: ~ - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- (string) Traces receiver address - address: ~ - # -- (string) Traces receiver port - port: ~ - # -- (string) Traces receiver credentials - creds: ~ - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- (string) Otel collector endpoint - collector: ~ - # -- (string) Otel collector credentials - creds: ~ - - # -- reportsController server port - # in case you are using hostNetwork: true, you might want to change the port the reportsController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - - # -- Enable sanity check for reports CRDs - sanityChecks: true diff --git a/helm-overrides/k8s-shared-int-ase1/kyverno/policies/protect-namespace.yaml b/helm-overrides/k8s-shared-int-ase1/kyverno/policies/protect-namespace.yaml deleted file mode 100644 index a028888..0000000 --- a/helm-overrides/k8s-shared-int-ase1/kyverno/policies/protect-namespace.yaml +++ /dev/null @@ -1,39 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: protect-critical-namespaces -spec: - validationFailureAction: Enforce - background: false - rules: - - name: deny-deletion-of-critical-namespaces - match: - any: - - resources: - kinds: - - Namespace - names: - - keda-shared-int - - contour-internal-0-shared-int - - contour-internal-1-shared-int - - contour-external-shared-int - - victoriametrics - - kube-system - - external-secrets-shared-int - - monitoring - - telegraf-operator - - fluentd - - kube-events - - opentelemetry - - contour-cert-checker-ns - - nginx-shared-internal - - contour-internal-1-shared-int-intra - - dg-self-hosted-int - validate: - message: "Deletion of critical namespaces is not allowed." - deny: - conditions: - any: - - key: "{{request.operation}}" - operator: Equals - value: DELETE diff --git a/helm-overrides/k8s-shared-int-ase1/kyverno/policies/request-limit-policy.yaml b/helm-overrides/k8s-shared-int-ase1/kyverno/policies/request-limit-policy.yaml deleted file mode 100644 index b92e2b4..0000000 --- a/helm-overrides/k8s-shared-int-ase1/kyverno/policies/request-limit-policy.yaml +++ /dev/null @@ -1,31 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: require-resource-requests-selected-ns -spec: - validationFailureAction: Enforce - background: false # does NOT touch existing resources - rules: - - name: require-cpu-memory-requests - match: - resources: - kinds: - - Deployment - - StatefulSet - namespaces: - # - check-ip - - "*" - operations: - - CREATE - - UPDATE - validate: - message: "CPU and memory requests must be specified for all containers." - pattern: - spec: - template: - spec: - containers: - - resources: - requests: - cpu: "?*" - memory: "?*" diff --git a/helm-overrides/k8s-shared-int-ase1/kyverno/policies/restrict-replicas.yaml b/helm-overrides/k8s-shared-int-ase1/kyverno/policies/restrict-replicas.yaml deleted file mode 100644 index 791f59b..0000000 --- a/helm-overrides/k8s-shared-int-ase1/kyverno/policies/restrict-replicas.yaml +++ /dev/null @@ -1,27 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: restrict-minimum-replicas-selected-ns -spec: - validationFailureAction: Enforce - background: false - rules: - - name: min-replicas-deployments-statefulsets - match: - resources: - kinds: - - Deployment - - StatefulSet - names: - - coredns - namespaces: - - kube-system - operations: - - CREATE - - UPDATE - validate: - message: "Deployments and StatefulSets must have at least 5 replicas in this namespace." - anyPattern: - # Case 1: replicas explicitly set and >= 5 - - spec: - replicas: ">=5" diff --git a/helm-overrides/k8s-shared-int-ase1/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index 1da6948..0000000 --- a/helm-overrides/k8s-shared-int-ase1/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,112 +0,0 @@ -fullnameOverride: opentelemetry-shared-int - -mode: daemonset - -priorityClassName: "system-node-critical" - -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: cloud.google.com/compute-class - operator: NotIn - values: - - cc-contour-vmagent-od - - cc-shared-devops-od - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a - -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 500m - memory: 256Mi - limits: - cpu: 1 - memory: 512Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "shared" - team: "sre" - service: "opentelemetry-shared-int" - env: "int" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/k8s-shared-int-ase1/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index 406a4b0..0000000 --- a/helm-overrides/k8s-shared-int-ase1/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,504 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 200m - # memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-admin-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "shared" - team: "shared-sre" - service: "node-exporter-shared-int" - env: "int" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -kuName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-shared-int-ase1/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index 95a1b33..0000000 --- a/helm-overrides/k8s-shared-int-ase1/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,177 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-shared-int" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-shared-int" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "shared" - team: "shared-sre" - service: "stackdriver-exporter-shared-int" - env: "int" - priority: "p0" - type: "exporter" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-shared-int-0622" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,kubernetes.io/node/ephemeral_storage/used_bytes,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - cloud.google.com/compute-class: cc-shared-devops-od - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: cloud.google.com/compute-class - operator: "Equal" - value: "cc-shared-devops-od" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-shrd-srsre-stackdriver-int@meesho-shared-int-0525.iam.gserviceaccount.com - } - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/k8s-shared-int-ase1/telegraf-operator/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/telegraf-operator/custom-values.yaml deleted file mode 100644 index 4dbdeb0..0000000 --- a/helm-overrides/k8s-shared-int-ase1/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,240 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - - infra-v2: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - namepass = ["DOWNSTREAM"] - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - namedrop = ["DOWNSTREAM"] - stats = ["count"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - expiration_interval = "2m" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "shared" - team: "shared-sre" - service: "telegraf-operator-shared-int" - env: "int" - priority: "p0" - type: "exporter" - -nodeSelector: - cloud.google.com/compute-class: "cc-shared-devops-od" - -tolerations: - - key: cloud.google.com/compute-class - operator: "Equal" - value: "cc-shared-devops-od" - effect: "NoSchedule" - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false \ No newline at end of file diff --git a/helm-overrides/k8s-shared-int-ase1/victoria-metrics-agent/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index d6296f3..0000000 --- a/helm-overrides/k8s-shared-int-ase1/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-shared-int -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-shared-int-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-shrd-srsre-vmagent-int@meesho-shared-int-0525.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-int-shared.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-infra-prd.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "shared" - team: "shared-sre" - service: "vmagent-shared-int" - env: "int" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-shared-int"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: metricsapi-shared-int.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 6 - memory: 4Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - cloud.google.com/compute-class: "cc-contour-vmagent-od" - -tolerations: - - key: cloud.google.com/compute-class - operator: "Equal" - value: "cc-contour-vmagent-od" - effect: "NoSchedule" - - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-shared-int-ase1/victoriametrics-agent/custom-values.yaml b/helm-overrides/k8s-shared-int-ase1/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index 2d308f7..0000000 --- a/helm-overrides/k8s-shared-int-ase1/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,474 +0,0 @@ - -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "shared-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -# -- Replica count -replicaCount: 2 - -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - -# -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables -deployment: - spec: - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - minReadySeconds: 300 - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - -image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - # -- Image tag, set to `Chart.AppVersion` by default - tag: v1.133.0 # rewrites Chart.AppVersion - # -- Variant of the image to use. - # e.g. enterprise, scratch - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - -# -- Image pull secrets -imagePullSecrets: [] - -# -- Add additional DNS entries to pods hosts file. Check [official documentation](https://kubernetes.io/docs/tasks/network/customize-hosts-file-for-pods/) -hostAliases: [] -# - ip: 192.168.1.1 -# hostNames: -# - test.example.com -# - another.example.net - -# -- Override chart name -nameOverride: "" - -# -- Override resources fullname -fullnameOverride: "vm-agent-shared-int" - -# -- Container working directory -containerWorkingDir: "/" - -rbac: - # -- Enables Role/RoleBinding creation - create: true - - # -- Role/RoleBinding annotations - annotations: {} - - # -- Role/RoleBinding labels - extraLabels: {} - - # -- If true and `rbac.enabled`, will deploy a Role/RoleBinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - - # -- additional rules for a role - extraRules: [] - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - # -- Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-shrd-srsre-vmagent-int@meesho-shared-int-0525.iam.gserviceaccount.com - } - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # -- mount API token to pod directly - automountToken: true - -# -- See `kubectl explain poddisruptionbudget.spec` for more or check [official documentation](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# -- Generates `remoteWrite.*` flags and config maps with value content for values, that are of type list of map. -# Each item should contain `url` param to pass validation. -remoteWrite: - - url: http://vm-insert-infra-prd.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/api/v1/write -# urlRelabelConfig: -# - action: keep -# source_labels: [env] -# regex: "dev" - -# -- VMAgent extra command line arguments -extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8429 - promscrape.config.strictParse: false # default is true - promscrape.maxScrapeSize: 1000000000 # default is 167772160 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - # promscrape.suppressScrapeErrors: true # default is false - - # Uncomment and specify the port if you want to support any of the protocols: - # https://docs.victoriametrics.com/victoriametrics/vmagent/#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for more details. -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth-secret - # key: password - -# -- Specify alternative source for env variables -envFrom: - [] - #- configMapRef: - # name: special-config - -# -- Extra labels for Deployment and Statefulset -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "shared" - team: "shared-sre" - service: "vm-agent-shared-int" - env: "int" - priority: "p0" - type: "vmagent" - -podLabels: - bu: "shared" - team: "shared-sre" - service: "vm-agent-shared-int" - env: "int" - priority: "p0" - type: "vmagent" - - -# -- Extra selector labels common for pod and service -selectorLabels: {} - -# -- Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# -- Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# -- Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -# -- Extra containers to run in a pod with vmagent -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -# -- Init containers for vmagent -initContainers: - [] - # - name: example - # image: example-image - -# -- Security context to be added to pod -podSecurityContext: - enabled: true - # fsGroup: 2000 - -# -- Security context to be added to pod's containers -securityContext: - enabled: true - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - # -- Enable agent service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - extraLabels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) for details - externalIPs: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8429 - # -- Target port - targetPort: http - # nodePort: 30000 - # -- Service type - type: ClusterIP - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - -ingress: - # -- Enable deployment of ingress for agent - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: metricsapi-shared-int.meeshogcp.in - path: - - / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - - # -- Ingress controller class name - ingressClassName: "nginx-internal" - - # -- Ingress path type - pathType: Prefix - -# -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) -resources: - requests: - cpu: 6 - memory: 50Gi - -# -- Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - cloud.google.com/compute-class: "cc-contour-vmagent-od" - -tolerations: - - key: cloud.google.com/compute-class - operator: "Equal" - value: "cc-contour-vmagent-od" - effect: "NoSchedule" - - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: topology.kubernetes.io/zone - operator: In - values: - - asia-southeast1-a - -# -- Pod affinity -affinity: {} - -# -- VMAgent [scraping configuration](https://docs.victoriametrics.com/victoriametrics/vmagent/#how-to-collect-metrics-in-prometheus-format) -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-shared-int" - -# -- Priority class to be assigned to the pod(s) -priorityClassName: "" - -# -- Enable the host network -hostNetwork: false - -serviceMonitor: - # -- Enable deployment of Service Monitor for server component. This is Prometheus operator object - enabled: false - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Service Monitor relabelings - relabelings: [] - # -- Basic auth params for Service Monitor - basicAuth: {} - # -- Service Monitor metricRelabelings - metricRelabelings: [] - # -- Service Monitor targetPort - targetPort: http - # interval: 15s - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - -# -- Empty dir configuration for a case, when persistence is disabled -emptyDir: {} - -persistentVolume: - # -- Create/use Persistent Volume Claim for server component. Empty dir if false - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Size of the volume. Should be calculated based on the logs you send and retention policy you set. - size: 10Gi - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume additional labels - extraLabels: {} - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Horizontal Pod Autoscaling. -# Note that it is not intended to be used for vmagents which perform scraping. -# In order to scale scraping vmagents check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) -horizontalPodAutoscaling: - # -- Use HPA for vmagent - enabled: false - # -- Maximum replicas for HPA to use to to scale vmagent - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale vmagent - minReplicas: 1 - # -- Metric for HPA to use to scale vmagent - metrics: [] - -# -- VMAgent scrape configuration -config: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -probe: - # -- Readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - # -- Liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - # -- Startup probe - startup: {} - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -allowedMetricsEndpoints: - - /metrics - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-supply-dev-ase1/deepgram-onprem/custom-values.yaml b/helm-overrides/k8s-supply-dev-ase1/deepgram-onprem/custom-values.yaml deleted file mode 100644 index d112911..0000000 --- a/helm-overrides/k8s-supply-dev-ase1/deepgram-onprem/custom-values.yaml +++ /dev/null @@ -1,795 +0,0 @@ -global: - # -- (string) If using images from the Deepgram Quay image repositories, - # or another private registry to which your cluster doesn't have default access, - # you will need to provide a pre-configured K8s Secret - # with image repository credentials. See chart docs for more details. - pullSecretRef: dg-regcred - - # -- (string) Name of the pre-configured K8s Secret containing your Deepgram - # self-hosted API key. See chart docs for more details. - deepgramSecretRef: dg-self-hosted-api-key - - # -- Additional labels to add to all Deepgram resources - additionalLabels: {} - - # -- When an API or Engine container is signaled to shutdown via Kubernetes sending a SIGTERM - # signal, the container will stop listening on its port, and no new requests will be routed - # to that container. However, the container will continue to run until all existing - # batch or streaming requests have completed, after which it will gracefully shut down. - # - # Batch requests should be finished within 10-15 minutes, but streaming requests can proceed indefinitely. - # - # outstandingRequestGracePeriod defines the period (in sec) after which Kubernetes will forcefully - # shutdown the container, terminating any outstanding connections. 1800 / 60 sec/min = 30 mins - outstandingRequestGracePeriod: 1800 - -# -- Configuration options for horizontal scaling of Deepgram -# services. Only one of `static` and `auto` options can be enabled. -# @default -- `` -scaling: - # -- Number of replicas to set during initial installation. - # @default -- `` - replicas: - api: 1 - engine: 1 - - # -- Enable pod autoscaling based on system load/traffic. - # @default -- `` - auto: - enabled: false - - api: - metrics: - # -- Scale the API deployment to this Engine-to-Api pod ratio - engineToApiRatio: 4 - # -- (list) If you have custom metrics you would like to scale with, you may add them here. - # See the [k8s docs](https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/) - # for how to structure a list of metrics - custom: - - # -- [Configurable scaling behavior](https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#configurable-scaling-behavior) - # @default -- "*See values.yaml file for default*" - behavior: - scaleDown: - policies: - - type: Pods - value: 1 - periodSeconds: 60 - - type: Percent - value: 25 - periodSeconds: 60 - - engine: - # -- Minimum number of Engine replicas. - minReplicas: 1 - # -- Maximum number of Engine replicas. - maxReplicas: 10 - metrics: - # -- If `engine.concurrencyLimit.activeRequests` is set, this variable will - # define the ratio of current active requests to maximum active requests at which - # the Engine pods will scale. Setting this value too close to 1.0 may lead to a situation where - # the cluster is at max capacity and rejects incoming requests. Setting the ratio too close to 0.0 - # will over-optimistically scale your cluster and increase compute costs unnecessarily. - requestCapacityRatio: - speechToText: - batch: - # -- (int) Scale the Engine pods based on a static desired number of speech-to-text batch requests per pod - requestsPerPod: - streaming: - # -- (int) Scale the Engine pods based on a static desired number of speech-to-text streaming requests per pod - requestsPerPod: - textToSpeech: - batch: - # -- (int) Scale the Engine pods based on a static desired number of text-to-speech batch requests per pod - requestsPerPod: - # -- If you have custom metrics you would like to scale with, you may add them here. - # See the [k8s docs](https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/) - # for how to structure a list of metrics - custom: [] - - # -- [Configurable scaling behavior](https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#configurable-scaling-behavior) - # @default -- "*See values.yaml file for default*" - behavior: - scaleDown: - policies: - - type: Pods - value: 1 - periodSeconds: 60 - - type: Percent - value: 25 - periodSeconds: 60 - -api: - # -- namePrefix is the prefix to apply to the name of all K8s objects - # associated with the Deepgram API containers. - namePrefix: "deepgram-api" - - image: - # -- path configures the image path to use for creating API containers. - # You may change this from the public Quay image path if you have imported - # Deepgram images into a private container registry. - path: quay.io/deepgram/self-hosted-api - # -- pullPolicy configures how the Kubelet attempts to pull the Deepgram API image - pullPolicy: IfNotPresent - # -- tag defines which Deepgram release to use for API containers - tag: release-240827 - - # -- Additional labels to add to API resources - additionalLabels: {} - - # -- (object) Additional annotations to add to the API deployment - additionalAnnotations: - - updateStrategy: - rollingUpdate: - # -- The maximum number of API pods, relative to the number of replicas, - # that can go offline during a rolling update. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-unavailable) - # for more details. - maxUnavailable: 0 - # -- The maximum number of extra API pods that can be created during a rollingUpdate, - # relative to the number of replicas. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-surge) - # for more details. - maxSurge: 1 - - # -- Configure resource limits per API container. See - # [Deepgram's documentation](https://developers.deepgram.com/docs/self-hosted-deployment-environments#api) - # for more details. - # @default -- `` - resources: - requests: - memory: "4Gi" - cpu: "2000m" - limits: - memory: "8Gi" - cpu: "4000m" - - # -- Readiness probe customization for API pods. - # @default -- `` - readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 1 - # -- Liveness probe customization for API pods. - # @default -- `` - livenessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 3 - - # -- [Affinity and anti-affinity](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#affinity-and-anti-affinity) - # to apply for API pods. - affinity: {} - # -- [Tolerations](https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/) - # to apply to API pods. - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepgram-api-pool - - - # -- [Security context](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) for API pods. - securityContext: {} - - serviceAccount: - # -- Specifies whether to create a default service account for the Deepgram API Deployment. - create: true - # -- (string) Allows providing a custom service account name for the API component. - # If left empty, the default service account name will be used. - # If specified, and `api.serviceAccount.create = true`, this defines the name of the default service account. - # If specified, and `api.serviceAccount.create = false`, this provides the name of a preconfigured service account - # you wish to attach to the API deployment. - name: - - # -- Configure how the API will listen for your requests - # @default -- `` - server: - # baseUrl is the prefix requests to the API. - baseUrl: "/v1" - # -- host is the IP address to listen on. You will want to listen - # on all interfaces to interact with other pods in the cluster. - host: "0.0.0.0" - # -- port to listen on. - port: 8080 - - # -- callbackConnTimeout configures how long to wait for a connection to a callback URL. - # See [Deepgram's callback documentation](https://developers.deepgram.com/docs/callback) - # for more details. The value should be a humantime duration. - callbackConnTimeout: "1s" - # -- callbackTimeout configures how long to wait for a response from a callback URL. - # See [Deepgram's callback documentation](https://developers.deepgram.com/docs/callback) - # for more details. The value should be a humantime duration. - callbackTimeout: "10s" - - # -- fetchConnTimeout configures how long to wait for a connection to a fetch URL. - # The value should be a humantime duration. - # A fetch URL is a URL passed in an inference request from which a payload should be - # downloaded. - fetchConnTimeout: "1s" - # -- fetchTimeout configures how long to wait for a response from a fetch URL. - # The value should be a humantime duration. - # A fetch URL is a URL passed in an inference request from which a payload should be - # downloaded. - fetchTimeout: "60s" - - # -- Specify custom DNS resolution options. - # @default -- `` - resolver: - # -- nameservers allows for specifying custom domain name server(s). - # A valid list item's format is "{IP} {PORT} {PROTOCOL (tcp or udp)}", - # e.g. `"127.0.0.1 53 udp"`. - nameservers: [] - # -- (int) maxTTL sets the DNS TTL value if specifying a custom DNS nameserver. - maxTTL: - - # -- Enable ancillary features - # @default -- `` - features: - # -- Enables entity detection on pre-recorded audio - # *if* a valid entity detection model is available. - # *WARNING*: Beta functionality. - entityDetection: false - - # -- Enables entity-based redaction on pre-recorded audio - # *if* a valid entity detection model is available. - # *WARNING*: Beta functionality. - entityRedaction: false - - # -- If API is receiving requests faster than Engine can process them, a request - # queue will form. By default, this queue is stored in memory. Under high load, - # the queue may grow too large and cause Out-Of-Memory errors. To avoid this, - # set a diskBufferPath to buffer the overflow on the request queue to disk. - # - # WARN: This is only to temporarily buffer requests during high load. - # If there is not enough Engine capacity to process the queued requests over time, - # the queue (and response time) will grow indefinitely. - diskBufferPath: - - # -- driverPool configures the backend pool of speech engines (generically referred to as - # "drivers" here). The API will load-balance among drivers in the standard - # pool; if one standard driver fails, the next one will be tried. - # @default -- `` - driverPool: - # -- standard is the main driver pool to use. - # @default -- `` - standard: - # -- timeoutBackoff is the factor to increase the timeout by - # for each additional retry (for exponential backoff). - timeoutBackoff: 1.2 - - # -- retrySleep defines the initial sleep period (in humantime duration) - # before attempting a retry. - retrySleep: "2s" - # -- retryBackoff is the factor to increase the retrySleep - # by for each additional retry (for exponential backoff). - retryBackoff: 1.6 - - # -- Maximum response to deserialize from Driver (in bytes). - # Default is 1GB, expressed in bytes. - maxResponseSize: "1073741824" - -engine: - # -- namePrefix is the prefix to apply to the name of all K8s objects - # associated with the Deepgram Engine containers. - namePrefix: "deepgram-engine" - - image: - # -- path configures the image path to use for creating Engine containers. - # You may change this from the public Quay image path if you have imported - # Deepgram images into a private container registry. - path: quay.io/deepgram/self-hosted-engine - # -- pullPolicy configures how the Kubelet attempts to pull the Deepgram Engine image - pullPolicy: IfNotPresent - # -- tag defines which Deepgram release to use for Engine containers - tag: release-240827 - - # -- Additional labels to add to Engine resources - additionalLabels: {} - - # -- (object) Additional annotations to add to the Engine deployment - additionalAnnotations: - - updateStrategy: - rollingUpdate: - # -- The maximum number of Engine pods, relative to the number of replicas, - # that can go offline during a rolling update. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-unavailable) - # for more details. - maxUnavailable: 0 - # -- The maximum number of extra Engine pods that can be created during a rollingUpdate, - # relative to the number of replicas. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-surge) - # for more details. - maxSurge: 1 - - # -- Configure resource limits per Engine container. See - # [Deepgram's documentation](https://developers.deepgram.com/docs/self-hosted-deployment-environments#engine) - # for more details. - # @default -- `` - resources: - requests: - memory: "30Gi" - cpu: "4000m" - # -- gpu maps to the nvidia.com/gpu resource parameter - gpu: 1 - limits: - memory: "40Gi" - cpu: "8000m" - # -- gpu maps to the nvidia.com/gpu resource parameter - gpu: 1 - - # -- The startupProbe combination of `periodSeconds` and `failureThreshold` allows - # time for the container to load all models and start listening for incoming requests. - # - # Model load time can be affected by hardware I/O speeds, as well as network speeds - # if you are using a network volume mount for the models. - # - # If you are hitting the failure threshold before models are finished loading, you may - # want to extend the startup probe. However, this will also extend the time it takes - # to detect a pod that can't establish a network connection to validate its license. - # @default -- `` - startupProbe: - # -- periodSeconds defines how often to execute the probe. - periodSeconds: 10 - # -- failureThreshold defines how many unsuccessful startup probe attempts - # are allowed before the container will be marked as Failed - failureThreshold: 60 - - # -- Readiness probe customization for Engine pods. - # @default -- `` - readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 1 - # -- Liveness probe customization for Engine pods. - # @default -- `` - livenessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 3 - - # -- [Affinity and anti-affinity](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#affinity-and-anti-affinity) - # to apply for Engine pods. - affinity: {} - # -- [Tolerations](https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/) - # to apply to Engine pods. - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dg-eg-pool - - # -- [Security context](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) for API pods. - securityContext: {} - - serviceAccount: - # -- Specifies whether to create a default service account for the Deepgram Engine Deployment. - create: true - # -- (string) Allows providing a custom service account name for the Engine component. - # If left empty, the default service account name will be used. - # If specified, and `engine.serviceAccount.create = true`, this defines the name of the default service account. - # If specified, and `engine.serviceAccount.create = false`, this provides the name of a preconfigured service account - # you wish to attach to the Engine deployment. - name: - - concurrencyLimit: - # -- (int) activeRequests limits the number of active requests handled by - # a single Engine container. - # If additional requests beyond the limit are sent, the API container - # forming the request will try a different Engine pod. If no Engine pods - # are able to accept the request, the API will return a 429 HTTP response - # to the client. The `nil` default means no limit will be set. - activeRequests: - - # -- Configure Engine containers to listen for requests from API containers. - # @default -- `` - server: - # -- host is the IP address to listen on for inference requests. - # You will want to listen on all interfaces to interact with - # other pods in the cluster. - host: "0.0.0.0" - # -- port to listen on for inference requests - port: 8080 - - # -- metricsServer exposes an endpoint on each Engine container - # for reporting inference-specific system metrics. - # See https://developers.deepgram.com/docs/metrics-guide#deepgram-engine - # for more details. - # @default -- `` - metricsServer: - # -- host is the IP address to listen on for metrics requests. - # You will want to listen on all interfaces to interact with - # other pods in the cluster. - host: "0.0.0.0" - # -- port to listen on for metrics requests - port: 9991 - - modelManager: - volumes: - customVolumeClaim: - # -- You may manually create your own PersistentVolume and PersistentVolumeClaim to store and - # expose model files to the Deepgram Engine. Configure your storage beforehand, - # and enable here. - # Note: Make sure the PV and PVC accessMode are set to `readWriteMany` or `readOnlyMany` - enabled: false - # -- (string) Name of your pre-configured PersistentVolumeClaim - name: - # -- Name of the directory within your pre-configured PersistentVolume - # where the models are stored - modelsDirectory: "/" - - aws: - efs: - # -- Whether to use an [AWS Elastic File Sytem](https://aws.amazon.com/efs/) - # to store Deepgram models for use by Engine containers. - # This option requires your cluster to be running in - # [AWS EKS](https://aws.amazon.com/eks/). - enabled: false - # -- Name prefix for the resources associated with the model storage in AWS EFS. - namePrefix: dg-models - # -- (string) FileSystemId of existing AWS Elastic File System where - # Deepgram model files will be persisted. - # You can find it using the AWS CLI: - # ``` - # $ aws efs describe-file-systems --query "FileSystems[*].FileSystemId" - # ``` - fileSystemId: - # -- Whether to force a fresh download of all model links provided, - # even if models are already present in EFS. - forceDownload: false - gcp: - gpd: - # -- Whether to use an [GKE Persistent Disks](https://cloud.google.com/kubernetes-engine/docs/concepts/persistent-volumes) - # to store Deepgram models for use by Engine containers. - # This option requires your cluster to be running in - # [GCP GKE](https://cloud.google.com/kubernetes-engine). - # See the GKE documentation on - # [using pre-existing persistent disks](https://cloud.google.com/kubernetes-engine/docs/how-to/persistent-volumes/preexisting-pd). - enabled: true - # -- Name prefix for the resources associated with the model storage in GCP GPD. - namePrefix: dg-models - # -- The storageClassName of the existing persistent disk. - storageClassName: "standard-rwo" - # -- The size of your pre-existing persistent disk. - storageCapacity: "40G" - # -- The identifier of your pre-existing persistent disk. - # The format is projects/{project_id}/zones/{zone_name}/disks/{disk_name} for Zonal persistent disks, - # or projects/{project_id}/regions/{region_name}/disks/{disk_name} for Regional persistent disks. - volumeHandle: "projects/meesho-supply-prd-0622/zones/asia-southeast1-a/disks/deepgram-model-storage" - fsType: "ext4" - - models: - # -- Links to your Deepgram models, if automatically downloading - # into storage backing a persistent volume. - # **Automatic downloads are currently supported for AWS EFS volumes only.** - # Insert each model link provided to you by your Deepgram - # Account Representative. - links: [] - - # -- chunking defines the size of audio chunks to process in seconds. - # Adjusting these values will affect both inference performance and accuracy - # of results. Please contact your Deepgram Account Representative if you - # want to adjust any of these values. - # @default -- `` - chunking: - speechToText: - batch: - # -- (float) minDuration is the minimum audio duration for a STT chunk size for a batch request - minDuration: - # -- (float) minDuration is the maximum audio duration for a STT chunk size for a batch request - maxDuration: - streaming: - # -- (float) minDuration is the minimum audio duration for a STT chunk size for a streaming request - minDuration: - # -- (float) minDuration is the maximum audio duration for a STT chunk size for a streaming request - maxDuration: - # -- step defines how often to return interim results, in seconds. - # This value may be lowered to increase the frequency of interim results. - # However, this also causes a significant decrease in the number of concurrent - # streams supported by a single GPU. Please contact your Deepgram Account - # representative for more details. - step: 0.2 - - halfPrecision: - # -- Engine will automatically enable half precision operations if your GPU supports - # them. You can explicitly enable or disable this behavior with the state parameter - # which supports `"enable"`, `"disabled"`, and `"auto"`. - state: "auto" - -# -- Configuration options for the optional -# [Deepgram License Proxy](https://developers.deepgram.com/docs/license-proxy). -# @default -- `` -licenseProxy: - # -- The License Proxy is optional, but highly recommended to be deployed in production - # to enable highly available environments. - enabled: true - - # -- If the License Proxy is deployed, one replica should be sufficient to - # support many API/Engine pods. - # Highly available environments may wish to deploy a second replica to ensure - # uptime, which can be toggled with this option. - deploySecondReplica: false - - # -- Even with a License Proxy deployed, API and Engine pods can be configured to keep the - # upstream `license.deepgram.com` license server as a fallback licensing option if the - # License Proxy is unavailable. - # Disable this option if you are restricting API/Engine Pod network access for security reasons, - # and only the License Proxy should send egress traffic to the upstream license server. - keepUpstreamServerAsBackup: true - - # -- namePrefix is the prefix to apply to the name of all K8s objects - # associated with the Deepgram License Proxy containers. - namePrefix: "deepgram-license-proxy" - - image: - # -- path configures the image path to use for creating License Proxy containers. - # You may change this from the public Quay image path if you have imported - # Deepgram images into a private container registry. - path: quay.io/deepgram/self-hosted-license-proxy - # -- tag defines which Deepgram release to use for License Proxy containers - tag: release-240827 - # -- pullPolicy configures how the Kubelet attempts to pull the Deepgram - # License Proxy image - pullPolicy: IfNotPresent - - # -- Additional labels to add to License Proxy resources - additionalLabels: {} - - # -- (object) Additional annotations to add to the LicenseProxy deployment - additionalAnnotations: - - updateStrategy: - # -- For the LicenseProxy, we only expose maxSurge and not maxUnavailable. - # This is to avoid accidentally having all LicenseProxy nodes go offline during upgrades, - # which could impact the entire cluster's connection to the Deepgram License Server. - # @default -- `` - rollingUpdate: - # -- The maximum number of extra License Proxy pods that can be created during a rollingUpdate, - # relative to the number of replicas. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-surge) - # for more details. - maxSurge: 1 - - # -- Configure resource limits per License Proxy container. See - # [Deepgram's documentation](https://developers.deepgram.com/docs/license-proxy#system-requirements) - # for more details. - # @default -- `` - resources: - requests: - memory: "1Gi" - cpu: "1000m" - limits: - memory: "8Gi" - cpu: "2000m" - - # -- Readiness probe customization for License Proxy pods. - # @default -- `` - readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 1 - # -- Liveness probe customization for Proxy pods. - # @default -- `` - livenessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 3 - - # -- [Affinity and anti-affinity](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#affinity-and-anti-affinity) - # to apply for License Proxy pods. - affinity: {} - # -- [Tolerations](https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/) - # to apply to License Proxy pods. - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepgram-proxy-pool - - # -- [Security context](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) for API pods. - securityContext: {} - - serviceAccount: - # -- Specifies whether to create a default service account for the Deepgram License Proxy Deployment. - create: true - # -- (string) Allows providing a custom service account name for the LicenseProxy component. - # If left empty, the default service account name will be used. - # If specified, and `licenseProxy.serviceAccount.create = true`, this defines the name of the default service account. - # If specified, and `licenseProxy.serviceAccount.create = false`, this provides the name of a preconfigured service account - # you wish to attach to the License Proxy deployment. - name: - - # -- Configure how the license proxy will listen for licensing requests. - # @default -- `` - server: - # --host is the IP address to listen on. You will want to listen - # on all interfaces to interact with other pods in the cluster. - host: "0.0.0.0" - # -- port to listen on. - port: 8443 - - # -- baseUrl is the prefix for incoming license verification requests. - baseUrl: "/" - - # -- statusPort is the port to listen on for the status/health endpoint. - statusPort: 8080 - -# -- Passthrough values for [NVIDIA GPU Operator Helm chart](https://github.com/NVIDIA/gpu-operator/blob/master/deployments/gpu-operator/values.yaml) -# You may use the NVIDIA GPU Operator to manage installation of NVIDIA drivers and the container toolkit on nodes with attached GPUs. -# @default -- `` -gpu-operator: - # -- Whether to install the NVIDIA GPU Operator to manage driver and/or container toolkit installation. - # See the list of [supported Operating Systems](https://docs.nvidia.com/datacenter/cloud-native/gpu-operator/latest/platform-support.html#supported-operating-systems-and-kubernetes-platforms) - # to verify compatibility with your cluster/nodes. Disable this option if your cluster/nodes are not compatible. - # If disabled, you will need to self-manage NVIDIA software installation on all nodes where you want - # to schedule Deepgram Engine pods. - enabled: false - driver: - # -- Whether to install NVIDIA drivers on nodes where a NVIDIA GPU is detected. - # If your Kubernetes nodes run a base image that comes with NVIDIA drivers pre-configured, - # disable this option, but keep the parent `gpu-operator` and sibling `toolkit` - # options enabled. - enabled: true - # -- NVIDIA driver version to install. - version: "550.54.15" - toolkit: - # -- Whether to install NVIDIA drivers on nodes where a NVIDIA GPU is detected. - enabled: true - # -- NVIDIA container toolkit to install. The default `ubuntu` image tag for the - # toolkit requires a dynamic runtime link to a version of GLIBC that may not be - # present on nodes running older Linux distribution releases, such as Ubuntu 22.04. - # Therefore, we specify the `ubi8` image, which statically links the GLIBC library - # and avoids this issue. - version: v1.15.0-ubi8 - -cluster-autoscaler: - # -- Set to `true` to enable node autoscaling with AWS EKS. Note needed for GKE, as autoscaling is enabled by a - # [cli option on cluster creation](https://cloud.google.com/kubernetes-engine/docs/how-to/cluster-autoscaler#creating_a_cluster_with_autoscaling). - enabled: false - rbac: - serviceAccount: - # -- Name of the IAM Service Account with the [necessary permissions](https://github.com/kubernetes/autoscaler/blob/master/cluster-autoscaler/cloudprovider/aws/README.md#permissions) - name: cluster-autoscaler-sa - annotations: - # -- (string) Replace with the AWS Role ARN configured for the Cluster Autoscaler. - # See the [Deepgram AWS EKS guide](https://developers.deepgram.com/docs/aws-k8s#creating-a-cluster) - # or [Cluster Autoscaler AWS documentation](https://github.com/kubernetes/autoscaler/blob/master/cluster-autoscaler/cloudprovider/aws/README.md#permissions) - # for details. - eks.amazonaws.com/role-arn: - autoDiscovery: - # -- (string) Name of your AWS EKS cluster. Using the [Cluster Autoscaler](https://github.com/kubernetes/autoscaler) - # on AWS requires knowledge of certain cluster metadata. - clusterName: - # -- (string) Region of your AWS EKS cluster. Using the [Cluster Autoscaler](https://github.com/kubernetes/autoscaler) - # on AWS requires knowledge of certain cluster metadata. - awsRegion: - -# -- Passthrough values for [Prometheus k8s stack Helm chart](https://github.com/prometheus-community/helm-charts/tree/main/charts/kube-prometheus-stack). -# Prometheus (and its adapter) should be configured when scaling.auto is enabled. -# You may choose to use the installation/configuration bundled in this Helm chart, -# or you may configure an existing Prometheus installation in your cluster to expose -# the needed values. -# See source Helm chart for explanation of available values. Default values provided in this chart are used -# to provide pod autoscaling for Deepgram pods. -# @default -- `` -kube-prometheus-stack: - # -- (bool) Normally, this chart will be installed if `scaling.auto.enabled` is true. However, if you wish - # to manage the Prometheus adapter in your cluster on your own and not as part of the Deepgram Helm chart, - # you can force it to not be installed by setting this to `false`. - includeDependency: - - fullnameOverride: "dg-prometheus-stack" - prometheus: - prometheusSpec: - additionalScrapeConfigs: - - job_name: "dg_engine_metrics" - scrape_interval: "2s" - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - "{{ .Release.Namespace }}" - relabel_configs: - - source_labels: [__meta_kubernetes_service_name] - regex: "(.*)-metrics" - action: keep - - source_labels: [__meta_kubernetes_endpoint_port_name] - regex: "metrics" - action: keep - - source_labels: [__meta_kubernetes_namespace] - target_label: namespace - - source_labels: [__meta_kubernetes_service_name] - target_label: service - - source_labels: [__meta_kubernetes_pod_name] - target_label: pod - - prometheusOperator: - enabled: true - - alertmanager: - enabled: false - - grafana: - enabled: true - - nodeExporter: - enabled: false - - kube-state-metrics: - enabled: true - metricLabelsAllowlist: - - namespaces=[{{ .Release.Namespace }}],deployments=[app] - -# -- Passthrough values for [Prometheus Adapter Helm chart](https://github.com/prometheus-community/helm-charts/tree/main/charts/prometheus-adapter). -# Prometheus, and its adapter here, should be configured when scaling.auto is enabled. -# You may choose to use the installation/configuration bundled in this Helm chart, -# or you may configure an existing Prometheus installation in your cluster to expose -# the needed values. -# See source Helm chart for explanation of available values. Default values provided in this chart are used -# to provide pod autoscaling for Deepgram pods. -# @default -- `` -prometheus-adapter: - # -- Normally, this chart will be installed if `scaling.auto.enabled` is true. However, if you wish - # to manage the Prometheus adapter in your cluster on your own and not as part of the Deepgram Helm chart, - # you can force it to not be installed by setting this to `false`. - includeDependency: - prometheus: - url: http://dg-prometheus-stack-prometheus.{{ .Release.Namespace }}.svc - rules: - default: false - external: - - name: - as: "engine_active_requests_stt_streaming" - seriesQuery: 'engine_active_requests{kind="stream"}' - metricsQuery: 'avg(engine_active_requests{kind="stream"})' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_active_requests_stt_batch" - seriesQuery: 'engine_active_requests{kind="batch"}' - metricsQuery: 'avg(engine_active_requests{kind="batch"})' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_active_requests_tts_batch" - seriesQuery: 'engine_active_requests{kind="tts"}' - metricsQuery: 'avg(engine_active_requests{kind="tts"})' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_estimated_stream_capacity" - seriesQuery: 'engine_active_requests{kind="stream"}' - metricsQuery: 'avg_over_time((sum(engine_active_requests{kind="stream"}) / sum(engine_estimated_stream_capacity) * 100)[1m:1m])' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_requests_active_to_max_ratio" - seriesQuery: "engine_max_active_requests" - metricsQuery: "avg_over_time((sum(engine_active_requests) / sum(engine_max_active_requests) * 100)[1m:1m])" - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_to_api_pod_ratio" - seriesQuery: 'kube_deployment_labels{label_app="deepgram-engine"}' - metricsQuery: '(sum(kube_deployment_status_replicas and on(deployment) kube_deployment_labels{label_app="deepgram-engine"})) / (sum(kube_deployment_status_replicas and on(deployment) kube_deployment_labels{label_app="deepgram-api"}))' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } diff --git a/helm-overrides/k8s-supply-prd-ase1/README.md b/helm-overrides/k8s-supply-prd-ase1/README.md deleted file mode 100644 index 27aeadc..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/README.md +++ /dev/null @@ -1,3 +0,0 @@ -# Cluster-based Custom Values - -This folder contains the custom `values.yaml` files organized based on specific cluster names. Each subdirectory corresponds to a particular cluster and holds the configurations for the applications and tools deployed within that cluster. \ No newline at end of file diff --git a/helm-overrides/k8s-supply-prd-ase1/alloy/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/alloy/custom-values.yaml deleted file mode 100644 index 01d4d6c..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/alloy/custom-values.yaml +++ /dev/null @@ -1,53 +0,0 @@ -fullnameOverride: "alloy-supply-prd" - -alloy: - configMap: - configFile: supply.alloy - - clustering: - enabled: true - - extraPorts: - - name: "otlp-grpc" - port: 4317 - targetPort: 4317 - protocol: "TCP" - - name: "otlp-http" - port: 4318 - targetPort: 4318 - protocol: "TCP" - - resources: - requests: - cpu: 7 - memory: 55Gi - limits: - cpu: 7.5 - memory: 60Gi - -configReloader: - enabled: true - -serviceAccount: - annotations: { - iam.gke.io/gcp-service-account: sa-supl-sre-grafna-obs-stk-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - } - -controller: - # Must be one of 'daemonset', 'deployment', or 'statefulset'. - type: 'deployment' - - nodeSelector: - dedicated: "alloy" - tolerations: - - key: "dedicated" - operator: "Equal" - value: "alloy" - effect: "NoSchedule" - - autoscaling: - enabled: true - minReplicas: 1 - maxReplicas: 50 - targetCPUUtilizationPercentage: 80 - targetMemoryUtilizationPercentage: 60 \ No newline at end of file diff --git a/helm-overrides/k8s-supply-prd-ase1/aurva-dataplane/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/aurva-dataplane/custom-values.yaml deleted file mode 100644 index 5306181..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/aurva-dataplane/custom-values.yaml +++ /dev/null @@ -1,676 +0,0 @@ -## Aurva Data Plane -## Ref: https://github.com/aurva-io/aurva-charts.git - -postgresql: - enabled: true - fullnameOverride: "aurva-dataplane-database" - volumePermissions: - ## @param volumePermissions.enabled Enable init container that changes the owner and group of the persistent volume - ## - enabled: true - global: - storageClass: pd-standard-retain - postgresql: - auth: - postgresPassword: "aurva" - database: "controller" - # Add toleration to make sure where this postgres db pod should reside (Applicable for production workloads): For more detail ref: https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/ - primary: - extendedConfiguration: | - max_connections = 300 - - #PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: supply-devops - - # -- Select nodes to deploy which matches the following labels - nodeSelector: ##PLACEHOLDER## - dedicated: supply-devops - -# -- Provide a name in place of `aurva` -# namespaceOverride: aurva-dataplane - -########################################################## -# Global Configs -########################################################## -global: - aurva_controller: - enabled: true - aurva_fastdet: - enabled: false - aurva_pii_analyzer: - enabled: true - aurva_ocr: - enabled: true - aurva_collector: - enabled: true - - deploymentAnnotations: {} - - priorityClassName: "" - -########################################################## -# Aurva Controller -########################################################## -aurva_controller: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "supply" - team: "supply-devops" - service: "aurva-supply-prd" - env: "prd" - priority: "p0" - type: "aurva_controller" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 10 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - image: - # Image of the app container - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-controller/aurva-controller - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - - type: secret - name: aurva-controller-secrets - # - type: configmap - # name: proxy-datasource-config - - # -- Resources to be defined for pod - resources: - limits: - memory: 2.5Gi - cpu: 2 - requests: - memory: 2Gi - cpu: 1 - - aurvaFastdet: - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-fastdet/aurva-fastdet - tag: "v2.30.13" - pullPolicy: IfNotPresent - envFrom: [] - env: [] - resources: - limits: - cpu: 0.5 - memory: 512Mi - requests: - cpu: 0.5 - memory: 512Mi - - nodeSelector: ##PLACEHOLDER## - dedicated: supply-devops - - # -- Taint tolerations for nodes - ##PLACEHOLDER## - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: supply-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-controller-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - config: - #variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - COMMAND_URL: "command.aurva-prd.meeshogcp.in:80" - DEPLOYMENT_TYPE: "kubernetes" - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - FLUSHER_WORKER_POOL_SIZE: "10000" - FLUSHER_BATCH_SIZE: "30000" - UNIQUENESS_IDENTIFIER: "k8s-supply-prd-ase1" #Recommendation: should be equal to cluster name - PROVIDER_ACCOUNT_ID: "meesho-supply-prd-0622" # GCP PROJECT ID (not Number) - REGION: "asia-southeast1" #eg: asia-south1 - ENVIRONMENT: "prod" - OCR_ENABLED: "true" - MONITORING_ENABLED: "false" - HYBRID_ONLY_MODE: "false" - FORCE_TLS: "false" - FASTDET_FLAG : "false" - AADHAAR_ENHANCER: "0" - WORKSPACE_EVENT_TRACKING_ENABLED: "false" - ACCESS_IQ_ENABLED: "true" - GRPC_ENFORCE_ALPN_ENABLED: "false" - ENABLE_FASTDET: "true" - FASTDET_MAX_BATCH_SIZE: "100" - SCAN_UUID_ENABLED: "true" - HEARTBEAT_INTERVAL: "5m" - #pii data - PII_BUCKET_NAME: "gcs-infra-devop-aurva-supply-prd" - PII_LOG_BUCKET_REGION: "asia-southeast1" - PII_LOG_CRON: "*/5 * * * *" - ENABLE_PII_LOG: "true" - PII_EVIDENCE_UPLOAD_MAX_BLOCKING_TASKS: "50000" - PII_EVIDENCE_UPLOAD_MAX_CONCURRENT_TASKS: "500" - PII_EVIDENCE_MAX_CACHE_WEIGHT: "100" - QUOTA_CLEANUP_CRON: "0 0 * * *" - ENABLE_PII_QUOTA: "true" - MAX_PII_EVIDENCES_PER_KEY_PER_WINDOW: "3" - PII_QUOTA_SYNC_CRON: "*/2 * * * *" - QUOTA_WINDOW_HOURS: "24" - #constants - SKIP_NAMESPACES: "contour-internal-1-supply-prd,contour-internal-0-supply-prd,contour-external-supply-prd" - CLOUD_PROVIDER: "gcp" - LOG_ENV: "production" - RDS_SCANNER_AVAILABILITY : "false" - REDSHIFT_SCANNER_AVAILABILITY : "false" - S3_SCANNER_AVAILABILITY : "false" - DYNAMO_SCANNER_AVAILABILITY: "false" - DOCDB_SCANNER_AVAILABILITY: "false" - OPENSEARCH_SCANNER_AVAILABILITY: "false" - CLOUDSQL_SCANNER_AVAILABILITY: "true" - BIGQUERY_SCANNER_AVAILABILITY: "true" - AWS_SNAPSHOT_SCANNER_AVAILABILITY: "false" - CLOUDSTORAGE_SCANNER_AVAILABILITY: "true" - KEYSPACES_SCANNER_AVAILABILITY: "false" - ALLOYDB_SCANNER_AVAILABILITY: "true" - BIGTABLE_SCANNER_AVAILABILITY: "true" - GCP_BACKUP_AVAILABILITY: "true" - EGRESS_MODE_ONLY: "false" - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-controller-sa - # -- Annotations applied to created service account - annotations: - iam.gke.io/gcp-service-account: sa-supply-prd-aurva-controller@meesho-supply-prd-0622.iam.gserviceaccount.com -# eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - autoscaling: - enabled: true - minReplicas: 10 - maxReplicas: 15 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 - - -########################################################## -# Aurva OCR -########################################################## -aurva_ocr: - # -- Additional labels for aurva-controller - additionalLabels: - bu: "supply" - team: "supply-devops" - service: "aurva-supply-prd" - env: "prd" - priority: "p0" - type: "aurva_ocr" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 1 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-ocr/aurva-ocr - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-ocr: - type: secret - name: aurva-ocr-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 2Gi - cpu: 1 - requests: - memory: 2Gi - cpu: 1 - - # -- Select nodes to deploy which matches the following labels - - nodeSelector: ##PLACEHOLDER## - dedicated: supply-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: supply-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-ocr-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - OCR_TIME_LIMIT: "1" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-supply-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-ocr-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # iam.gke.io/gcp-service-account: service-account@gcp.iam.gserviceaccount.com - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva Collector -########################################################## -aurva_collector: - - # -- Additional labels for aurva-analyzer - additionalLabels: - bu: "supply" - team: "supply-devops" - service: "aurva-supply-prd" - env: "prd" - priority: "p0" - type: "aurva_collector" - - - # -- Annotations on aurva-analyzer - annotations: {} - # "key": "value" - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-collector/aurva-collector - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-controller: - type: secret - name: aurva-collector-secrets - - resources: - limits: - cpu: 200m - memory: 800Mi - requests: - cpu: 200m - memory: 512Mi - - podSecurityContext: {} - - securityContext: - privileged: true - capabilities: - add: - # For kernel v5.8 and above we don't need CAP_SYS_ADMIN or CAP_SYS_RESOURCE - # we just need CAP_BPF and CAP_PERFMON. This has been tested on our EKS node - # which is on kernel v5.10.x - # When SSL Tracing is required we need CAP_SYS_ADMIN and CAP_SYS_PTRACE - # on top of the previous capabilities - # So finally these are the 4 possible combinations for capabilies - # 1. Newer Kernels without SSL - # - BPF - # - PERFMON - # 2. Newer Kernels with SSL - - SYS_ADMIN - - SYS_PTRACE - # 3. Older Kernels without SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # 4. Older Kernels with SSL - # - SYS_ADMIN - # - SYS_RESOURCE - # - SYS_PTRACE - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - volumes: - - name: debugfs - mountPath: /sys/kernel/debug - hostPath: /sys/kernel/debug - - name: vmlinux - mountPath: /sys/kernel/btf/vmlinux - hostPath: /sys/kernel/btf/vmlinux - - name: procfs - mountPath: /host/proc - hostPath: /proc - - name: bpffs - mountPath: /sys/fs/bpf - hostPath: /sys/fs/bpf - - # -- Taint tolerations for nodes - tolerations: - # - effect: NoSchedule - # key: dedicated - # operator: Equal - # value: megatetra - - operator: Exists - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external - - alloy - - preprod-spot-16 - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-collector-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - # variables - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-supply-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - TRACE_INTERNAL_SVC: "true" - TRACE_INTERNAL_SVC_HTTP: "true" - INTERNAL_SVC_SAMPLE_INTERVAL: "10m" - LOGS_TTL: "1h" - TRACE_HTTP2: "true" - TRACE_SSL: "false" - TRACE_PSQL: "false" - TRACE_SQLSERVER: "false" - TRACE_MYSQL: "false" - TRACE_EGRESS: "true" - TRACE_GO_TLS: "false" - TRACE_ML_SERVICES: "false" - # constants - LOG_ENV: production - SENTRY_DSN: "https://fd6738e1ee4a9a9c1f079d09953b43b1@sentry.aurva.io/4" - MONITORING_ENABLED: "false" - ENABLE_INGRESS_INFORMER: "false" - ENABLE_SERVICE_INFORMER: "false" - ENABLE_ISTIO_INFORMER: "false" - EXCLUDED_PII_REGEX_TYPES: "ip_address,us_bank_number,us_driver_license,us_itin,us_passport,us_routing,us_mbi,ssn" - AGGREGATOR_MAX_CONNECTIONS: "1000" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-collector-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - -########################################################## -# Aurva PII Analyzer -########################################################## -aurva_pii_analyzer: - - # -- Additional labels for aurva-controller - additionalLabels: - bu: "supply" - team: "supply-devops" - service: "aurva-supply-prd" - env: "prd" - priority: "p0" - type: "aurva_pii_analyzer" - - # -- Annotations on aurva-controller - annotations: {} - # "key": "value" - - revisionHistoryLimit: 3 - - # -- no of replicas for aurva controller - replicas: 16 - - # -- Additional label added on pod which is used in Service's Label Selector - podLabels: {} - - # -- Additional Pod Annotations added on pod created by this Deployment - additionalPodAnnotations: {} - # "key": "value" - - # -- Secrets used to pull image - imagePullSecrets: "" - - ##PLACEHOLDER## - nodeSelector: - dedicated: supply-devops - - # Image of the app container - image: - repository: asia-south1-docker.pkg.dev/aurva-gcp/aurva-piianalyzer/aurva-piianalyzer - tag: "v3.20.3" - pullPolicy: IfNotPresent - - # Environment variables to be passed to the app container - env: [] - - # -- If want to mount Envs from configmap or secret - envFrom: - aurva-pii-analyzer: - type: secret - name: aurva-pii-analyzer-secrets - - # -- Resources to be defined for pod - resources: - limits: - memory: 5Gi - cpu: 4 - requests: - memory: 4Gi - cpu: 2 - - nodeSelector: ##PLACEHOLDER## - dedicated: supply-devops - - ##PLACEHOLDER## - # -- Taint tolerations for nodes - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: supply-devops - - # -- Pod affinity and pod anti-affinity allow you to specify rules about how pods should be placed relative to other pods. - affinity: - # nodeAffinity: - # requiredDuringSchedulingIgnoredDuringExecution: - # nodeSelectorTerms: - # - matchExpressions: - # - key: disktype - # operator: In - # values: - # - ssd - - # -- [DNS configuration] - dnsConfig: {} - # -- Alternative DNS policy for application controller pods - dnsPolicy: "ClusterFirst" - - secret: - name: "aurva-pii-analyzer-secrets" - # -- Additional Labels on secrets - additionalLabels: - # key: value - # -- Annotations on secrets - annotations: - # key: value - - config: - PG_USERNAME: "postgres" - PG_PASSWORD: "aurva" - PG_DBNAME: "controller" - SCHEDULER_TIME: "1" - SUPPORTED_REGION: "US" - COMPANY_ID: "65eeb832-67ba-40fb-b95a-30ca9eaa3409" - UNIQUENESS_IDENTIFIER: "k8s-supply-prd-ase1" - DEPLOYMENT_TYPE: "kubernetes" - MAX_HOURS_TO_KEEP_REGARDLESS_OF_ACK_STATUS: "2880" - - serviceAccount: - # -- Create a service account for the aurva controller - create: true - # -- Service account name - name: aurva-pii-analyzer-sa - # -- Annotations applied to created service account - annotations: - # eks.amazonaws.com/role-arn: arn:aws:iam:::role/ - # -- Labels applied to created service account - labels: {} - - autoscaling: - enabled: true - minReplicas: 1 - maxReplicas: 2 - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 70 - - type: Resource - resource: - name: memory - target: - type: Utilization - averageUtilization: 70 diff --git a/helm-overrides/k8s-supply-prd-ase1/cert-manager/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/cert-manager/custom-values.yaml deleted file mode 100644 index f7814aa..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/cert-manager/custom-values.yaml +++ /dev/null @@ -1,129 +0,0 @@ -global: - logLevel: 2 - rbac: - create: true - priorityClassName: "high-priority" -installCRDs: false - -crds: - enabled: true - keep: true - -# Cert-manager Controller -replicaCount: 1 -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-controller - tag: v1.20.1 - pullPolicy: IfNotPresent - -nodeSelector: - dedicated: supply-devops - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "supply-devops" - effect: "NoSchedule" - -resources: - requests: - cpu: 100m - memory: 128Mi - limits: - cpu: 500m - memory: 512Mi - -# Webhook Configuration -webhook: - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-webhook - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: supply-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "supply-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# CA Injector Configuration -cainjector: - enabled: true - replicaCount: 1 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-cainjector - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: supply-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "supply-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 50m - memory: 64Mi - limits: - cpu: 250m - memory: 256Mi - -# ACME Solver Configuration -acmesolver: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-acmesolver - tag: v1.20.1 - pullPolicy: IfNotPresent - -# Startup API Check -startupapicheck: - enabled: true - timeout: 1m - backoffLimit: 4 - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/devops/cert-manager/cert-manager-startupapicheck - tag: v1.20.1 - pullPolicy: IfNotPresent - - nodeSelector: - dedicated: supply-devops - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "supply-devops" - effect: "NoSchedule" - - resources: - requests: - cpu: 10m - memory: 32Mi - limits: - cpu: 50m - memory: 64Mi - -# Prometheus Monitoring -prometheus: - enabled: true - servicemonitor: - enabled: false - interval: 60s - scrapeTimeout: 30s - labels: - prometheus: cert-manager diff --git a/helm-overrides/k8s-supply-prd-ase1/conntrack-adjuster/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index c390837..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,20 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - max: 2097152 - hashsize: 524288 - sleepInterval: 30 - -nodeAffinity: - enabled: true - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: In - values: - - "contour-external" - - "contour-internal-0" - - "contour-internal-1" - - "contour-intra-0" - - "contour-intra-1" diff --git a/helm-overrides/k8s-supply-prd-ase1/contour-ca-issuer/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/contour-ca-issuer/custom-values.yaml deleted file mode 100644 index 133ac64..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/contour-ca-issuer/custom-values.yaml +++ /dev/null @@ -1,17 +0,0 @@ -# ClusterIssuer override for k8s-supply-prd-ase1 (prd supply cluster, ase1a zone). -# ClusterIssuer is a cluster-scoped resource — this file pins the issuer -# identity for this cluster so the name is auditable per-cluster. -# -# Must match the `issuerRef.name` in consuming Certificate CRs (see -# devops-helm-charts/2.0.0/templates/proxyless-grpc-cert.yaml). - -issuerName: contour-supply-prd-ca-issuer -rootCASecretName: contour-supply-ca - -externalSecret: - enabled: true - vaultPath: meesho/devops/contour/root-ca - refreshInterval: "0" - secretStoreRef: - name: vault-backend -namespace: cert-manager-supply-prd \ No newline at end of file diff --git a/helm-overrides/k8s-supply-prd-ase1/contour-cert-checker/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/contour-cert-checker/custom-values.yaml deleted file mode 100644 index f0d12d2..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=k8s-supply-prd-ase1"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - dedicated: supply-devops - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: supply-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/k8s-supply-prd-ase1/contour-external/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/contour-external/custom-values.yaml deleted file mode 100644 index 98dfb80..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/contour-external/custom-values.yaml +++ /dev/null @@ -1,86 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled - network: - num-trusted-hops: 1 -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: supply - team: supply-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 2048Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-shared - nodeSelector: - dedicated: contour-shared - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: supply - team: supply-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external - nodeSelector: - dedicated: contour-external - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "40" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 8Gi - service: - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-supply-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-supply-prd-ase1/contour-internal-0/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/contour-internal-0/custom-values.yaml deleted file mode 100644 index 0a63e7f..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,99 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: supply - team: supply-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 4Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-shared - nodeSelector: - dedicated: contour-shared - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - enabled: true - podLabels: - bu: supply - team: supply-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "55" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 12 - memory: 8Gi - service: - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int-0-supply-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-supply-prd-ase1/contour-internal-1/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/contour-internal-1/custom-values.yaml deleted file mode 100644 index fc06f5a..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,92 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: supply - team: supply-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 12 - memory: 22Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-intra-0 - nodeSelector: - dedicated: contour-intra-0 - service: - tcpLB: true - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: supply - team: supply-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 12 - memory: 6Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int-1-supply-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-supply-prd-ase1/contour-internal-intra-0/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index 6bdcdb7..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,97 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: supply - team: supply-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 4Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-shared - nodeSelector: - dedicated: contour-shared - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - enabled: true - podLabels: - bu: supply - team: supply-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-intra-0 - nodeSelector: - dedicated: contour-intra-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "55" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 12 - memory: 8Gi - service: - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-supply-prd-ase1/contour-internal-intra-1/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index 7b7f60c..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,90 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - tlsExistingSecret: "contourcert" - enabled: true - replicaCount: 3 - podLabels: - bu: supply - team: supply-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 4Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-shared - nodeSelector: - dedicated: contour-shared - service: - tcpLB: true - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - tlsExistingSecret: "envoycert" - enabled: true - podLabels: - bu: supply - team: supply-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-intra-1 - nodeSelector: - dedicated: contour-intra-1 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "40" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-supply-prd-ase1/coredns/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/coredns/custom-values.yaml deleted file mode 100644 index ba5b696..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/coredns/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -replicaCount: 30 -communicationType: "intra" - -labels: - bu: supply - team: supply-devops - env: prd - -clusterIP: 10.137.8.2 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "supply-devops" - effect: "NoSchedule" - -nodeSelector: - dedicated: supply-devops - kubernetes.io/os: linux diff --git a/helm-overrides/k8s-supply-prd-ase1/coroot-node-agent/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/coroot-node-agent/custom-values.yaml deleted file mode 100644 index b6cc324..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/coroot-node-agent/custom-values.yaml +++ /dev/null @@ -1,54 +0,0 @@ -fullnameOverride: "coroot-node-agent-supply-prd" -priorityClassName: "low-priority" - -scrape: "false" # disable scraping for now - -resources: - requests: - cpu: "300m" - memory: "1Gi" - limits: - cpu: "500m" - memory: "1.5Gi" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-supply-contour-ext - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-supply-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external - - alloy - - sumoduolite-op - - sumounolite \ No newline at end of file diff --git a/helm-overrides/k8s-supply-prd-ase1/deepgram-onprem-v2/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/deepgram-onprem-v2/custom-values.yaml deleted file mode 100644 index a647602..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/deepgram-onprem-v2/custom-values.yaml +++ /dev/null @@ -1,410 +0,0 @@ -global: - pullSecretRef: dg-regcred - deepgramSecretRef: dg-self-hosted-api-key-v2 - additionalLabels: {} - outstandingRequestGracePeriod: 1800 - -apiAutoscaling: - enabled: true - targetName: deepgram-api - maxReplicas: 500 - minReplicas: 2 - pollingInterval: 30 - scaledown: - policies: - - periodseconds: 60 - type: Pods - value: 2 - selectpolicy: Min - stabilizationWindowSeconds: 900 - scaleup: - policies: - - periodseconds: 15 - type: Pods - value: 2 - - periodseconds: 15 - type: Percent - value: 10 - selectpolicy: Max - stabilizationWindowSeconds: 60 - triggers: - - metadata: - metricName: engine_active_requests - query: sum(kube_deployment_status_replicas_available{namespace="dg-self-hosted-v2",deployment="deepgram-engine"}) - serverAddress: http://vmselect-supply-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus/ - threshold: "0.5" - type: prometheus - -engineAutoscaling: - enabled: true - targetName: deepgram-engine - maxReplicas: 500 - minReplicas: 2 - pollingInterval: 30 - scaledown: - policies: - - periodseconds: 60 - type: Pods - value: 2 - selectpolicy: Min - stabilizationWindowSeconds: 900 - scaleup: - policies: - - periodseconds: 15 - type: Pods - value: 2 - - periodseconds: 15 - type: Percent - value: 10 - selectpolicy: Max - stabilizationWindowSeconds: 60 - triggers: - - metadata: - metricName: engine_active_requests - query: sum(engine_active_requests{kind="stream",kubernetes_namespace="dg-self-hosted-v2"}) - serverAddress: http://vmselect-supply-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus/ - threshold: "5" - type: prometheus - -scaling: - replicas: - api: 15 - engine: 20 - auto: - enabled: false - api: - metrics: - engineToApiRatio: 4 - custom: - behavior: - scaleDown: - policies: - - type: Pods - value: 1 - periodSeconds: 60 - - type: Percent - value: 25 - periodSeconds: 60 - engine: - minReplicas: 1 - maxReplicas: 10 - metrics: - requestCapacityRatio: 0.8 - speechToText: - batch: - requestsPerPod: 12 - streaming: - requestsPerPod: 14 - textToSpeech: - batch: - requestsPerPod: 50 - custom: [] - behavior: - scaleDown: - policies: - - type: Pods - value: 1 - periodSeconds: 60 - - type: Percent - value: 25 - periodSeconds: 60 - -api: - createContourGateway: true - ingress: - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: 'false' - nginx.ingress.kubernetes.io/ssl-redirect: 'false' - enabled: true - hosts: - - host: deepgram-v2.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / - apiServiceName: deepgram-api-external - servicePort: 80 - ingressClassName: contour-internal-0 - servicePort: 80 - enableWebsocket: false - namePrefix: deepgram-api - namespace: dg-self-hosted-v2 - slowStart: - enabled: true - window: 60s - aggression: 0.5 - minPercent: 5 - - image: - path: quay.io/deepgram/self-hosted-api - pullPolicy: IfNotPresent - tag: release-251118 - - additionalLabels: {} - additionalAnnotations: - - updateStrategy: - rollingUpdate: - maxUnavailable: 0 - maxSurge: 1 - - resources: - requests: - memory: "4Gi" - cpu: "2000m" - limits: - memory: "8Gi" - cpu: "4000m" - - readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 1 - livenessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 3 - - affinity: {} - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepgram-api-pool - - securityContext: {} - - serviceAccount: - create: true - name: - - server: - baseUrl: "/v1" - host: "0.0.0.0" - port: 8080 - callbackConnTimeout: "1s" - callbackTimeout: "10s" - fetchConnTimeout: "1s" - fetchTimeout: "60s" - - resolver: - nameservers: [] - maxTTL: - - features: - entityDetection: false - entityRedaction: false - diskBufferPath: - - driverPool: - standard: - timeoutBackoff: 1.2 - retrySleep: "2s" - retryBackoff: 1.6 - maxResponseSize: "1073741824" - -engine: - namePrefix: "deepgram-engine" - - image: - path: quay.io/deepgram/self-hosted-engine - pullPolicy: IfNotPresent - tag: release-251118 - - additionalLabels: {} - additionalAnnotations: - - updateStrategy: - rollingUpdate: - maxUnavailable: 0 - maxSurge: 1 - - resources: - requests: - memory: "15Gi" - cpu: "6" - gpu: 1 - limits: - memory: "20Gi" - cpu: "6" - gpu: 1 - - startupProbe: - periodSeconds: 10 - failureThreshold: 60 - - readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 1 - livenessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 3 - - lifecycle: - preStop: - exec: - command: - - /bin/bash - - -c - - /bin/sleep 30 - - affinity: {} - nodeSelector: - dedicated: dg-eg-pool - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dg-eg-pool - - effect: NoSchedule - key: nvidia.com/gpu - operator: Equal - value: present - - effect: NoSchedule - key: nvidia.com/gpu - operator: Exists - - securityContext: {} - - serviceAccount: - create: true - name: - - concurrencyLimit: - activeRequests: 12 - - server: - host: "0.0.0.0" - port: 8080 - - metricsServer: - host: "0.0.0.0" - port: 9273 - - modelManager: - volumes: - customVolumeClaim: - enabled: false - name: - modelsDirectory: "/" - aws: - efs: - enabled: false - namePrefix: dg-models - fileSystemId: - forceDownload: false - nova3: - enabled: false - multilingual: - enabled: true - gcp: - gpd: - enabled: true - namePrefix: dg-models-v2 - storageClassName: "standard-rwo" - storageCapacity: "50G" - volumeHandle: "projects/meesho-supply-prd-0622/zones/asia-southeast1-a/disks/deepgram-model-storage-nova3-multilingual-v2" - fsType: "ext4" - models: - links: [] - - chunking: - speechToText: - batch: - minDuration: - maxDuration: - streaming: - minDuration: - maxDuration: - step: 0.2 - - halfPrecision: - state: "auto" - -licenseProxy: - enabled: false - deploySecondReplica: false - keepUpstreamServerAsBackup: true - namePrefix: "deepgram-license-proxy" - - image: - path: quay.io/deepgram/self-hosted-license-proxy - tag: release-251118 - pullPolicy: IfNotPresent - - additionalLabels: {} - additionalAnnotations: - - updateStrategy: - rollingUpdate: - maxSurge: 1 - - resources: - requests: - memory: "1Gi" - cpu: "1000m" - limits: - memory: "8Gi" - cpu: "2000m" - - readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 1 - livenessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 3 - - affinity: {} - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepgram-proxy-pool - - securityContext: {} - - serviceAccount: - create: true - name: - - server: - host: "0.0.0.0" - port: 8443 - baseUrl: "/" - statusPort: 8080 - -gpu-operator: - enabled: false - driver: - enabled: true - version: "550.54.15" - toolkit: - enabled: true - version: v1.15.0-ubi8 - -cluster-autoscaler: - enabled: false - -kube-prometheus-stack: - includeDependency: false - fullnameOverride: "dg-prometheus-stack" - prometheusOperator: - enabled: false - alertmanager: - enabled: false - grafana: - enabled: false - nodeExporter: - enabled: false - kube-state-metrics: - enabled: false - -prometheus-adapter: - includeDependency: - prometheus: - url: http://dg-prometheus-stack-prometheus.{{ .Release.Namespace }}.svc - rules: - default: false diff --git a/helm-overrides/k8s-supply-prd-ase1/deepgram-onprem/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/deepgram-onprem/custom-values.yaml deleted file mode 100644 index e4bbf0d..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/deepgram-onprem/custom-values.yaml +++ /dev/null @@ -1,918 +0,0 @@ -global: - # -- (string) If using images from the Deepgram Quay image repositories, - # or another private registry to which your cluster doesn't have default access, - # you will need to provide a pre-configured K8s Secret - # with image repository credentials. See chart docs for more details. - pullSecretRef: dg-regcred - - # -- (string) Name of the pre-configured K8s Secret containing your Deepgram - # self-hosted API key. See chart docs for more details. - deepgramSecretRef: dg-self-hosted-api-key-v2 - - # -- Additional labels to add to all Deepgram resources - additionalLabels: {} - - # -- When an API or Engine container is signaled to shutdown via Kubernetes sending a SIGTERM - # signal, the container will stop listening on its port, and no new requests will be routed - # to that container. However, the container will continue to run until all existing - # batch or streaming requests have completed, after which it will gracefully shut down. - # - # Batch requests should be finished within 10-15 minutes, but streaming requests can proceed indefinitely. - # - # outstandingRequestGracePeriod defines the period (in sec) after which Kubernetes will forcefully - # shutdown the container, terminating any outstanding connections. 1800 / 60 sec/min = 30 mins - outstandingRequestGracePeriod: 1800 - -# -- Configuration options for horizontal scaling of Deepgram -# services. Only one of `static` and `auto` options can be enabled. -# @default -- `` - -apiAutoscaling: - enabled: true - targetName: deepgram-api - maxReplicas: 500 - minReplicas: 224 - pollingInterval: 30 - scaledown: - policies: - - periodseconds: 60 - type: Pods - value: 2 - selectpolicy: Min - stabilizationWindowSeconds: 900 - scaleup: - policies: - - periodseconds: 15 - type: Pods - value: 2 - - periodseconds: 15 - type: Percent - value: 10 - selectpolicy: Max - stabilizationWindowSeconds: 60 - triggers: - - metadata: - metricName: engine_active_requests - query: sum(kube_deployment_status_replicas_available{namespace="dg-self-hosted",deployment="deepgram-engine"}) - serverAddress: http://vmselect-supply-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus/ - threshold: "0.5" - type: prometheus - - metadata: - desiredReplicas: "10" - end: "0 7 * * *" - start: "30 23 * * *" - timezone: "Asia/Kolkata" - type: cron - - - -engineAutoscaling: - enabled: true - targetName: deepgram-engine - maxReplicas: 500 - minReplicas: 112 - pollingInterval: 30 - scaledown: - policies: - - periodseconds: 60 - type: Pods - value: 2 - selectpolicy: Min - stabilizationWindowSeconds: 900 - scaleup: - policies: - - periodseconds: 15 - type: Pods - value: 2 - - periodseconds: 15 - type: Percent - value: 10 - selectpolicy: Max - stabilizationWindowSeconds: 60 - triggers: - - metadata: - metricName: engine_active_requests - query: sum(engine_active_requests{kind="stream",kubernetes_namespace="dg-self-hosted"}) - serverAddress: http://vmselect-supply-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus/ - threshold: "5" - type: prometheus - - metadata: - desiredReplicas: "5" - end: "0 7 * * *" - start: "30 23 * * *" - timezone: "Asia/Kolkata" - type: cron - -scaling: - # -- Number of replicas to set during initial installation. - # @default -- `` - replicas: - api: 15 - engine: 20 - - # -- Enable pod autoscaling based on system load/traffic. - # @default -- `` - auto: - enabled: false - - api: - metrics: - # -- Scale the API deployment to this Engine-to-Api pod ratio - engineToApiRatio: 4 - # -- (list) If you have custom metrics you would like to scale with, you may add them here. - # See the [k8s docs](https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/) - # for how to structure a list of metrics - custom: - - # -- [Configurable scaling behavior](https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#configurable-scaling-behavior) - # @default -- "*See values.yaml file for default*" - behavior: - scaleDown: - policies: - - type: Pods - value: 1 - periodSeconds: 60 - - type: Percent - value: 25 - periodSeconds: 60 - - engine: - # -- Minimum number of Engine replicas. - minReplicas: 1 - # -- Maximum number of Engine replicas. - maxReplicas: 10 - metrics: - # -- If `engine.concurrencyLimit.activeRequests` is set, this variable will - # define the ratio of current active requests to maximum active requests at which - # the Engine pods will scale. Setting this value too close to 1.0 may lead to a situation where - # the cluster is at max capacity and rejects incoming requests. Setting the ratio too close to 0.0 - # will over-optimistically scale your cluster and increase compute costs unnecessarily. - requestCapacityRatio: 0.8 - speechToText: - batch: - # -- (int) Scale the Engine pods based on a static desired number of speech-to-text batch requests per pod - requestsPerPod: 12 - streaming: - # -- (int) Scale the Engine pods based on a static desired number of speech-to-text streaming requests per pod - requestsPerPod: 14 - textToSpeech: - batch: - # -- (int) Scale the Engine pods based on a static desired number of text-to-speech batch requests per pod - requestsPerPod: 50 - # -- If you have custom metrics you would like to scale with, you may add them here. - # See the [k8s docs](https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/) - # for how to structure a list of metrics - custom: [] - - # -- [Configurable scaling behavior](https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#configurable-scaling-behavior) - # @default -- "*See values.yaml file for default*" - behavior: - scaleDown: - policies: - - type: Pods - value: 1 - periodSeconds: 60 - - type: Percent - value: 25 - periodSeconds: 60 - -api: - # -- namePrefix is the prefix to apply to the name of all K8s objects - # associated with the Deepgram API containers. - createContourGateway: true - ingress: - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: 'false' - nginx.ingress.kubernetes.io/ssl-redirect: 'false' - enabled: true - hosts: - - host: deepgram.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / - apiServiceName: deepgram-api-external - servicePort: 80 - ingressClassName: contour-internal-1 - servicePort: 80 - enableWebsocket: false - namePrefix: deepgram-api - namespace: dg-self-hosted - slowStart: - enabled: true - window: 60s - aggression: 0.5 - minPercent: 5 - - - image: - # -- path configures the image path to use for creating API containers. - # You may change this from the public Quay image path if you have imported - # Deepgram images into a private container registry. - path: quay.io/deepgram/self-hosted-api - # -- pullPolicy configures how the Kubelet attempts to pull the Deepgram API image - pullPolicy: IfNotPresent - # -- tag defines which Deepgram release to use for API containers - tag: release-251118 - - # -- Additional labels to add to API resources - additionalLabels: {} - - # -- (object) Additional annotations to add to the API deployment - additionalAnnotations: - - updateStrategy: - rollingUpdate: - # -- The maximum number of API pods, relative to the number of replicas, - # that can go offline during a rolling update. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-unavailable) - # for more details. - maxUnavailable: 0 - # -- The maximum number of extra API pods that can be created during a rollingUpdate, - # relative to the number of replicas. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-surge) - # for more details. - maxSurge: 1 - - # -- Configure resource limits per API container. See - # [Deepgram's documentation](https://developers.deepgram.com/docs/self-hosted-deployment-environments#api) - # for more details. - # @default -- `` - resources: - requests: - memory: "4Gi" - cpu: "2000m" - limits: - memory: "8Gi" - cpu: "4000m" - - # -- Readiness probe customization for API pods. - # @default -- `` - readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 1 - # -- Liveness probe customization for API pods. - # @default -- `` - livenessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 3 - - # -- [Affinity and anti-affinity](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#affinity-and-anti-affinity) - # to apply for API pods. - affinity: {} - # -- [Tolerations](https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/) - # to apply to API pods. - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepgram-api-pool - - - # -- [Security context](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) for API pods. - securityContext: {} - - serviceAccount: - # -- Specifies whether to create a default service account for the Deepgram API Deployment. - create: true - # -- (string) Allows providing a custom service account name for the API component. - # If left empty, the default service account name will be used. - # If specified, and `api.serviceAccount.create = true`, this defines the name of the default service account. - # If specified, and `api.serviceAccount.create = false`, this provides the name of a preconfigured service account - # you wish to attach to the API deployment. - name: - - # -- Configure how the API will listen for your requests - # @default -- `` - server: - # baseUrl is the prefix requests to the API. - baseUrl: "/v1" - # -- host is the IP address to listen on. You will want to listen - # on all interfaces to interact with other pods in the cluster. - host: "0.0.0.0" - # -- port to listen on. - port: 8080 - - # -- callbackConnTimeout configures how long to wait for a connection to a callback URL. - # See [Deepgram's callback documentation](https://developers.deepgram.com/docs/callback) - # for more details. The value should be a humantime duration. - callbackConnTimeout: "1s" - # -- callbackTimeout configures how long to wait for a response from a callback URL. - # See [Deepgram's callback documentation](https://developers.deepgram.com/docs/callback) - # for more details. The value should be a humantime duration. - callbackTimeout: "10s" - - # -- fetchConnTimeout configures how long to wait for a connection to a fetch URL. - # The value should be a humantime duration. - # A fetch URL is a URL passed in an inference request from which a payload should be - # downloaded. - fetchConnTimeout: "1s" - # -- fetchTimeout configures how long to wait for a response from a fetch URL. - # The value should be a humantime duration. - # A fetch URL is a URL passed in an inference request from which a payload should be - # downloaded. - fetchTimeout: "60s" - - # -- Specify custom DNS resolution options. - # @default -- `` - resolver: - # -- nameservers allows for specifying custom domain name server(s). - # A valid list item's format is "{IP} {PORT} {PROTOCOL (tcp or udp)}", - # e.g. `"127.0.0.1 53 udp"`. - nameservers: [] - # -- (int) maxTTL sets the DNS TTL value if specifying a custom DNS nameserver. - maxTTL: - - # -- Enable ancillary features - # @default -- `` - features: - # -- Enables entity detection on pre-recorded audio - # *if* a valid entity detection model is available. - # *WARNING*: Beta functionality. - entityDetection: false - - # -- Enables entity-based redaction on pre-recorded audio - # *if* a valid entity detection model is available. - # *WARNING*: Beta functionality. - entityRedaction: false - - # -- If API is receiving requests faster than Engine can process them, a request - # queue will form. By default, this queue is stored in memory. Under high load, - # the queue may grow too large and cause Out-Of-Memory errors. To avoid this, - # set a diskBufferPath to buffer the overflow on the request queue to disk. - # - # WARN: This is only to temporarily buffer requests during high load. - # If there is not enough Engine capacity to process the queued requests over time, - # the queue (and response time) will grow indefinitely. - diskBufferPath: - - # -- driverPool configures the backend pool of speech engines (generically referred to as - # "drivers" here). The API will load-balance among drivers in the standard - # pool; if one standard driver fails, the next one will be tried. - # @default -- `` - driverPool: - # -- standard is the main driver pool to use. - # @default -- `` - standard: - # -- timeoutBackoff is the factor to increase the timeout by - # for each additional retry (for exponential backoff). - timeoutBackoff: 1.2 - - # -- retrySleep defines the initial sleep period (in humantime duration) - # before attempting a retry. - retrySleep: "2s" - # -- retryBackoff is the factor to increase the retrySleep - # by for each additional retry (for exponential backoff). - retryBackoff: 1.6 - - # -- Maximum response to deserialize from Driver (in bytes). - # Default is 1GB, expressed in bytes. - maxResponseSize: "1073741824" - -engine: - # -- namePrefix is the prefix to apply to the name of all K8s objects - # associated with the Deepgram Engine containers. - namePrefix: "deepgram-engine" - - image: - # -- path configures the image path to use for creating Engine containers. - # You may change this from the public Quay image path if you have imported - # Deepgram images into a private container registry. - path: quay.io/deepgram/self-hosted-engine - # -- pullPolicy configures how the Kubelet attempts to pull the Deepgram Engine image - pullPolicy: IfNotPresent - # -- tag defines which Deepgram release to use for Engine containers - tag: release-251118 - - # -- Additional labels to add to Engine resources - additionalLabels: {} - - # -- (object) Additional annotations to add to the Engine deployment - additionalAnnotations: - - updateStrategy: - rollingUpdate: - # -- The maximum number of Engine pods, relative to the number of replicas, - # that can go offline during a rolling update. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-unavailable) - # for more details. - maxUnavailable: 0 - # -- The maximum number of extra Engine pods that can be created during a rollingUpdate, - # relative to the number of replicas. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-surge) - # for more details. - maxSurge: 1 - - # -- Configure resource limits per Engine container. See - # [Deepgram's documentation](https://developers.deepgram.com/docs/self-hosted-deployment-environments#engine) - # for more details. - # @default -- `` - resources: - requests: - memory: "15Gi" - cpu: "6" - # -- gpu maps to the nvidia.com/gpu resource parameter - gpu: 1 - limits: - memory: "20Gi" - cpu: "6" - # -- gpu maps to the nvidia.com/gpu resource parameter - gpu: 1 - - # -- The startupProbe combination of `periodSeconds` and `failureThreshold` allows - # time for the container to load all models and start listening for incoming requests. - # - # Model load time can be affected by hardware I/O speeds, as well as network speeds - # if you are using a network volume mount for the models. - # - # If you are hitting the failure threshold before models are finished loading, you may - # want to extend the startup probe. However, this will also extend the time it takes - # to detect a pod that can't establish a network connection to validate its license. - # @default -- `` - startupProbe: - # -- periodSeconds defines how often to execute the probe. - periodSeconds: 10 - # -- failureThreshold defines how many unsuccessful startup probe attempts - # are allowed before the container will be marked as Failed - failureThreshold: 60 - - # -- Readiness probe customization for Engine pods. - # @default -- `` - readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 1 - # -- Liveness probe customization for Engine pods. - # @default -- `` - livenessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 3 - - # -- [Container lifecycle hooks](https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/) - lifecycle: - preStop: - exec: - command: - - /bin/bash - - -c - - /bin/sleep 30 - - # -- [Affinity and anti-affinity](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#affinity-and-anti-affinity) - # to apply for Engine pods. - affinity: {} - # -- [Tolerations](https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/) - # to apply to Engine pods. - nodeSelector: - dedicated: dg-eg-pool - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: dg-eg-pool - - effect: NoSchedule - key: nvidia.com/gpu - operator: Equal - value: present - - effect: NoSchedule - key: nvidia.com/gpu - operator: Exists - - - # -- [Security context](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) for API pods. - securityContext: {} - - serviceAccount: - # -- Specifies whether to create a default service account for the Deepgram Engine Deployment. - create: true - # -- (string) Allows providing a custom service account name for the Engine component. - # If left empty, the default service account name will be used. - # If specified, and `engine.serviceAccount.create = true`, this defines the name of the default service account. - # If specified, and `engine.serviceAccount.create = false`, this provides the name of a preconfigured service account - # you wish to attach to the Engine deployment. - name: - - concurrencyLimit: - # -- (int) activeRequests limits the number of active requests handled by - # a single Engine container. - # If additional requests beyond the limit are sent, the API container - # forming the request will try a different Engine pod. If no Engine pods - # are able to accept the request, the API will return a 429 HTTP response - # to the client. The `nil` default means no limit will be set. - activeRequests: 12 - - # -- Configure Engine containers to listen for requests from API containers. - # @default -- `` - server: - # -- host is the IP address to listen on for inference requests. - # You will want to listen on all interfaces to interact with - # other pods in the cluster. - host: "0.0.0.0" - # -- port to listen on for inference requests - port: 8080 - - # -- metricsServer exposes an endpoint on each Engine container - # for reporting inference-specific system metrics. - # See https://developers.deepgram.com/docs/metrics-guide#deepgram-engine - # for more details. - # @default -- `` - metricsServer: - # -- host is the IP address to listen on for metrics requests. - # You will want to listen on all interfaces to interact with - # other pods in the cluster. - host: "0.0.0.0" - # -- port to listen on for metrics requests - port: 9273 - - modelManager: - volumes: - customVolumeClaim: - # -- You may manually create your own PersistentVolume and PersistentVolumeClaim to store and - # expose model files to the Deepgram Engine. Configure your storage beforehand, - # and enable here. - # Note: Make sure the PV and PVC accessMode are set to `readWriteMany` or `readOnlyMany` - enabled: false - # -- (string) Name of your pre-configured PersistentVolumeClaim - name: - # -- Name of the directory within your pre-configured PersistentVolume - # where the models are stored - modelsDirectory: "/" - - aws: - efs: - # -- Whether to use an [AWS Elastic File Sytem](https://aws.amazon.com/efs/) - # to store Deepgram models for use by Engine containers. - # This option requires your cluster to be running in - # [AWS EKS](https://aws.amazon.com/eks/). - enabled: false - # -- Name prefix for the resources associated with the model storage in AWS EFS. - namePrefix: dg-models - # -- (string) FileSystemId of existing AWS Elastic File System where - # Deepgram model files will be persisted. - # You can find it using the AWS CLI: - # ``` - # $ aws efs describe-file-systems --query "FileSystems[*].FileSystemId" - # ``` - fileSystemId: - # -- Whether to force a fresh download of all model links provided, - # even if models are already present in EFS. - forceDownload: false - nova3: - enabled: false - multilingual: - enabled: true - gcp: - gpd: - # -- Whether to use an [GKE Persistent Disks](https://cloud.google.com/kubernetes-engine/docs/concepts/persistent-volumes) - # to store Deepgram models for use by Engine containers. - # This option requires your cluster to be running in - # [GCP GKE](https://cloud.google.com/kubernetes-engine). - # See the GKE documentation on - # [using pre-existing persistent disks](https://cloud.google.com/kubernetes-engine/docs/how-to/persistent-volumes/preexisting-pd). - enabled: true - # -- Name prefix for the resources associated with the model storage in GCP GPD. - namePrefix: dg-models - # -- The storageClassName of the existing persistent disk. - storageClassName: "standard-rwo" - # -- The size of your pre-existing persistent disk. - storageCapacity: "50G" - # -- The identifier of your pre-existing persistent disk. - # The format is projects/{project_id}/zones/{zone_name}/disks/{disk_name} for Zonal persistent disks, - # or projects/{project_id}/regions/{region_name}/disks/{disk_name} for Regional persistent disks. - volumeHandle: "projects/meesho-supply-prd-0622/zones/asia-southeast1-a/disks/deepgram-model-storage-nova3-multilingual" - fsType: "ext4" - - models: - # -- Links to your Deepgram models, if automatically downloading - # into storage backing a persistent volume. - # **Automatic downloads are currently supported for AWS EFS volumes only.** - # Insert each model link provided to you by your Deepgram - # Account Representative. - links: [] - - # -- chunking defines the size of audio chunks to process in seconds. - # Adjusting these values will affect both inference performance and accuracy - # of results. Please contact your Deepgram Account Representative if you - # want to adjust any of these values. - # @default -- `` - chunking: - speechToText: - batch: - # -- (float) minDuration is the minimum audio duration for a STT chunk size for a batch request - minDuration: - # -- (float) minDuration is the maximum audio duration for a STT chunk size for a batch request - maxDuration: - streaming: - # -- (float) minDuration is the minimum audio duration for a STT chunk size for a streaming request - minDuration: - # -- (float) minDuration is the maximum audio duration for a STT chunk size for a streaming request - maxDuration: - # -- step defines how often to return interim results, in seconds. - # This value may be lowered to increase the frequency of interim results. - # However, this also causes a significant decrease in the number of concurrent - # streams supported by a single GPU. Please contact your Deepgram Account - # representative for more details. - step: 0.2 - - halfPrecision: - # -- Engine will automatically enable half precision operations if your GPU supports - # them. You can explicitly enable or disable this behavior with the state parameter - # which supports `"enable"`, `"disabled"`, and `"auto"`. - state: "auto" - -# -- Configuration options for the optional -# [Deepgram License Proxy](https://developers.deepgram.com/docs/license-proxy). -# @default -- `` -licenseProxy: - # -- The License Proxy is optional, but highly recommended to be deployed in production - # to enable highly available environments. - enabled: false - - # -- If the License Proxy is deployed, one replica should be sufficient to - # support many API/Engine pods. - # Highly available environments may wish to deploy a second replica to ensure - # uptime, which can be toggled with this option. - deploySecondReplica: false - - # -- Even with a License Proxy deployed, API and Engine pods can be configured to keep the - # upstream `license.deepgram.com` license server as a fallback licensing option if the - # License Proxy is unavailable. - # Disable this option if you are restricting API/Engine Pod network access for security reasons, - # and only the License Proxy should send egress traffic to the upstream license server. - keepUpstreamServerAsBackup: true - - # -- namePrefix is the prefix to apply to the name of all K8s objects - # associated with the Deepgram License Proxy containers. - namePrefix: "deepgram-license-proxy" - - image: - # -- path configures the image path to use for creating License Proxy containers. - # You may change this from the public Quay image path if you have imported - # Deepgram images into a private container registry. - path: quay.io/deepgram/self-hosted-license-proxy - # -- tag defines which Deepgram release to use for License Proxy containers - tag: release-251118 - # -- pullPolicy configures how the Kubelet attempts to pull the Deepgram - # License Proxy image - pullPolicy: IfNotPresent - - # -- Additional labels to add to License Proxy resources - additionalLabels: {} - - # -- (object) Additional annotations to add to the LicenseProxy deployment - additionalAnnotations: - - updateStrategy: - # -- For the LicenseProxy, we only expose maxSurge and not maxUnavailable. - # This is to avoid accidentally having all LicenseProxy nodes go offline during upgrades, - # which could impact the entire cluster's connection to the Deepgram License Server. - # @default -- `` - rollingUpdate: - # -- The maximum number of extra License Proxy pods that can be created during a rollingUpdate, - # relative to the number of replicas. See the - # [Kubernetes documentation](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#max-surge) - # for more details. - maxSurge: 1 - - # -- Configure resource limits per License Proxy container. See - # [Deepgram's documentation](https://developers.deepgram.com/docs/license-proxy#system-requirements) - # for more details. - # @default -- `` - resources: - requests: - memory: "1Gi" - cpu: "1000m" - limits: - memory: "8Gi" - cpu: "2000m" - - # -- Readiness probe customization for License Proxy pods. - # @default -- `` - readinessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 1 - # -- Liveness probe customization for Proxy pods. - # @default -- `` - livenessProbe: - initialDelaySeconds: 5 - periodSeconds: 10 - failureThreshold: 3 - - # -- [Affinity and anti-affinity](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#affinity-and-anti-affinity) - # to apply for License Proxy pods. - affinity: {} - # -- [Tolerations](https://kubernetes.io/docs/concepts/scheduling-eviction/taint-and-toleration/) - # to apply to License Proxy pods. - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: deepgram-proxy-pool - - # -- [Security context](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) for API pods. - securityContext: {} - - serviceAccount: - # -- Specifies whether to create a default service account for the Deepgram License Proxy Deployment. - create: true - # -- (string) Allows providing a custom service account name for the LicenseProxy component. - # If left empty, the default service account name will be used. - # If specified, and `licenseProxy.serviceAccount.create = true`, this defines the name of the default service account. - # If specified, and `licenseProxy.serviceAccount.create = false`, this provides the name of a preconfigured service account - # you wish to attach to the License Proxy deployment. - name: - - # -- Configure how the license proxy will listen for licensing requests. - # @default -- `` - server: - # --host is the IP address to listen on. You will want to listen - # on all interfaces to interact with other pods in the cluster. - host: "0.0.0.0" - # -- port to listen on. - port: 8443 - - # -- baseUrl is the prefix for incoming license verification requests. - baseUrl: "/" - - # -- statusPort is the port to listen on for the status/health endpoint. - statusPort: 8080 - -# -- Passthrough values for [NVIDIA GPU Operator Helm chart](https://github.com/NVIDIA/gpu-operator/blob/master/deployments/gpu-operator/values.yaml) -# You may use the NVIDIA GPU Operator to manage installation of NVIDIA drivers and the container toolkit on nodes with attached GPUs. -# @default -- `` -gpu-operator: - # -- Whether to install the NVIDIA GPU Operator to manage driver and/or container toolkit installation. - # See the list of [supported Operating Systems](https://docs.nvidia.com/datacenter/cloud-native/gpu-operator/latest/platform-support.html#supported-operating-systems-and-kubernetes-platforms) - # to verify compatibility with your cluster/nodes. Disable this option if your cluster/nodes are not compatible. - # If disabled, you will need to self-manage NVIDIA software installation on all nodes where you want - # to schedule Deepgram Engine pods. - enabled: false - driver: - # -- Whether to install NVIDIA drivers on nodes where a NVIDIA GPU is detected. - # If your Kubernetes nodes run a base image that comes with NVIDIA drivers pre-configured, - # disable this option, but keep the parent `gpu-operator` and sibling `toolkit` - # options enabled. - enabled: true - # -- NVIDIA driver version to install. - version: "550.54.15" - toolkit: - # -- Whether to install NVIDIA drivers on nodes where a NVIDIA GPU is detected. - enabled: true - # -- NVIDIA container toolkit to install. The default `ubuntu` image tag for the - # toolkit requires a dynamic runtime link to a version of GLIBC that may not be - # present on nodes running older Linux distribution releases, such as Ubuntu 22.04. - # Therefore, we specify the `ubi8` image, which statically links the GLIBC library - # and avoids this issue. - version: v1.15.0-ubi8 - -cluster-autoscaler: - # -- Set to `true` to enable node autoscaling with AWS EKS. Note needed for GKE, as autoscaling is enabled by a - # [cli option on cluster creation](https://cloud.google.com/kubernetes-engine/docs/how-to/cluster-autoscaler#creating_a_cluster_with_autoscaling). - enabled: false - rbac: - serviceAccount: - # -- Name of the IAM Service Account with the [necessary permissions](https://github.com/kubernetes/autoscaler/blob/master/cluster-autoscaler/cloudprovider/aws/README.md#permissions) - name: cluster-autoscaler-sa - annotations: - # -- (string) Replace with the AWS Role ARN configured for the Cluster Autoscaler. - # See the [Deepgram AWS EKS guide](https://developers.deepgram.com/docs/aws-k8s#creating-a-cluster) - # or [Cluster Autoscaler AWS documentation](https://github.com/kubernetes/autoscaler/blob/master/cluster-autoscaler/cloudprovider/aws/README.md#permissions) - # for details. - eks.amazonaws.com/role-arn: - autoDiscovery: - # -- (string) Name of your AWS EKS cluster. Using the [Cluster Autoscaler](https://github.com/kubernetes/autoscaler) - # on AWS requires knowledge of certain cluster metadata. - clusterName: - # -- (string) Region of your AWS EKS cluster. Using the [Cluster Autoscaler](https://github.com/kubernetes/autoscaler) - # on AWS requires knowledge of certain cluster metadata. - awsRegion: - -# -- Passthrough values for [Prometheus k8s stack Helm chart](https://github.com/prometheus-community/helm-charts/tree/main/charts/kube-prometheus-stack). -# Prometheus (and its adapter) should be configured when scaling.auto is enabled. -# You may choose to use the installation/configuration bundled in this Helm chart, -# or you may configure an existing Prometheus installation in your cluster to expose -# the needed values. -# See source Helm chart for explanation of available values. Default values provided in this chart are used -# to provide pod autoscaling for Deepgram pods. -# @default -- `` -kube-prometheus-stack: - # -- (bool) Normally, this chart will be installed if `scaling.auto.enabled` is true. However, if you wish - # to manage the Prometheus adapter in your cluster on your own and not as part of the Deepgram Helm chart, - # you can force it to not be installed by setting this to `false`. - includeDependency: false - - fullnameOverride: "dg-prometheus-stack" - prometheus: - prometheusSpec: - additionalScrapeConfigs: - - job_name: "dg_engine_metrics" - scrape_interval: "2s" - kubernetes_sd_configs: - - role: endpoints - namespaces: - names: - - "{{ .Release.Namespace }}" - relabel_configs: - - source_labels: [__meta_kubernetes_service_name] - regex: "(.*)-metrics" - action: keep - - source_labels: [__meta_kubernetes_endpoint_port_name] - regex: "metrics" - action: keep - - source_labels: [__meta_kubernetes_namespace] - target_label: namespace - - source_labels: [__meta_kubernetes_service_name] - target_label: service - - source_labels: [__meta_kubernetes_pod_name] - target_label: pod - - prometheusOperator: - enabled: false - - alertmanager: - enabled: false - - grafana: - enabled: false - - nodeExporter: - enabled: false - - kube-state-metrics: - enabled: false - metricLabelsAllowlist: - - namespaces=[{{ .Release.Namespace }}],deployments=[app] - -# -- Passthrough values for [Prometheus Adapter Helm chart](https://github.com/prometheus-community/helm-charts/tree/main/charts/prometheus-adapter). -# Prometheus, and its adapter here, should be configured when scaling.auto is enabled. -# You may choose to use the installation/configuration bundled in this Helm chart, -# or you may configure an existing Prometheus installation in your cluster to expose -# the needed values. -# See source Helm chart for explanation of available values. Default values provided in this chart are used -# to provide pod autoscaling for Deepgram pods. -# @default -- `` -prometheus-adapter: - # -- Normally, this chart will be installed if `scaling.auto.enabled` is true. However, if you wish - # to manage the Prometheus adapter in your cluster on your own and not as part of the Deepgram Helm chart, - # you can force it to not be installed by setting this to `false`. - includeDependency: - prometheus: - url: http://dg-prometheus-stack-prometheus.{{ .Release.Namespace }}.svc - rules: - default: false - external: - - name: - as: "engine_active_requests_stt_streaming" - seriesQuery: 'engine_active_requests{kind="stream"}' - metricsQuery: 'avg(engine_active_requests{kind="stream"})' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_active_requests_stt_batch" - seriesQuery: 'engine_active_requests{kind="batch"}' - metricsQuery: 'avg(engine_active_requests{kind="batch"})' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_active_requests_tts_batch" - seriesQuery: 'engine_active_requests{kind="tts"}' - metricsQuery: 'avg(engine_active_requests{kind="tts"})' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_estimated_stream_capacity" - seriesQuery: 'engine_active_requests{kind="stream"}' - metricsQuery: 'avg_over_time((sum(engine_active_requests{kind="stream"}) / sum(engine_estimated_stream_capacity) * 100)[1m:1m])' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_requests_active_to_max_ratio" - seriesQuery: "engine_max_active_requests" - metricsQuery: "avg_over_time((sum(engine_active_requests) / sum(engine_max_active_requests) * 100)[1m:1m])" - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } - service: { resource: "service" } - - name: - as: "engine_to_api_pod_ratio" - seriesQuery: 'kube_deployment_labels{label_app="deepgram-engine"}' - metricsQuery: '(sum(kube_deployment_status_replicas and on(deployment) kube_deployment_labels{label_app="deepgram-engine"})) / (sum(kube_deployment_status_replicas and on(deployment) kube_deployment_labels{label_app="deepgram-api"}))' - resources: - overrides: - namespace: { resource: "namespace" } - pod: { resource: "pod" } diff --git a/helm-overrides/k8s-supply-prd-ase1/external-secrets/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/external-secrets/custom-values.yaml deleted file mode 100644 index 93b6cbb..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/external-secrets/custom-values.yaml +++ /dev/null @@ -1,61 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Annotations to add to Pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: supply-devops - nodeSelector: - dedicated: supply-devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: supply-devops - nodeSelector: - dedicated: supply-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: supply-devops - nodeSelector: - dedicated: supply-devops - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - prometheus.io/path: "/metrics" \ No newline at end of file diff --git a/helm-overrides/k8s-supply-prd-ase1/flagger/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/flagger/custom-values.yaml deleted file mode 100644 index a063869..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/flagger/custom-values.yaml +++ /dev/null @@ -1,63 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - dedicated: supply-devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: supply-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: supply-devops - bu: infra - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-supply-rollout-service.prd-supply-rollout-service.svc.cluster.local" - -rolloutService: - enabled: true - apps: - \ No newline at end of file diff --git a/helm-overrides/k8s-supply-prd-ase1/fluentd-sumoduolite-np/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/fluentd-sumoduolite-np/custom-values.yaml deleted file mode 100644 index 8dd9811..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/fluentd-sumoduolite-np/custom-values.yaml +++ /dev/null @@ -1,724 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-supl-spsre-fluentd-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 1000m - memory: 500Mi - limits: - memory: 700Mi - cpu: 1200m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: In - values: - - sumoduolite-op - - sumounolite -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: supply - team: sre - type: fluentd - service: fluentd-supply-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: supply - team: sre - type: fluentd - service: fluentd-supply-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: [] -# - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-supply-prd-ase1/fluentd/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/fluentd/custom-values.yaml deleted file mode 100644 index 71e68a0..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/fluentd/custom-values.yaml +++ /dev/null @@ -1,855 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-supl-spsre-fluentd-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - sumoduolite-op - - sumounolite -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: supply - team: sre - type: fluentd - service: fluentd-supply-prd - priority: p0 - env: prd - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: supply - team: sre - type: fluentd - service: fluentd-supply-prd - priority: p0 - env: prd - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey -- secretRef: - name: es-password - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: - - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - from_encoding "utf-8" - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @DG_SELF_HOSTED - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - - - - - - - - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-supply-prd-ase1/keda/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/keda/custom-values.yaml deleted file mode 100644 index a8beaeb..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/keda/custom-values.yaml +++ /dev/null @@ -1,45 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: supply-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "supply-devops" - effect: "NoSchedule" - podLabels: - bu: "supply" - team: "supply-devops" - metricsAdapter: - bu: "supply" - team: "supply-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 2000Mi - requests: - cpu: 700m - memory: 1000Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 120m - memory: 300Mi - metricServer: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 120m - memory: 200Mi - env: - - name: KEDA_SCALEDOBJECT_CTRL_MAX_RECONCILES - value: '25' diff --git a/helm-overrides/k8s-supply-prd-ase1/kube-dns/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/kube-dns/custom-values.yaml deleted file mode 100644 index 95404a9..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.137.8.2"],"prd.mrouter.int.svc.cluster.local":["10.137.8.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.137.8.2"]} diff --git a/helm-overrides/k8s-supply-prd-ase1/kube-events/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/kube-events/custom-values.yaml deleted file mode 100644 index 418954f..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-supply-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - dedicated: supply-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: supply-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: supply-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: supply-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: supply-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: supply-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-supply-prd-ase1/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 566ae88..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-supply-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-supply-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "supply" - team: "supply-sre" - service: "kube-state-metrics-supply-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "vmselect-mds" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 10m - memory: 50Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-supply-prd-ase1/kubectl-mcp-server/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/kubectl-mcp-server/custom-values.yaml deleted file mode 100644 index b818b5f..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/kubectl-mcp-server/custom-values.yaml +++ /dev/null @@ -1,117 +0,0 @@ -fullnameOverride: "kubectl-mcp-server" - -replicas: 1 - -image: - # In-house hardened build (helm-templates/kubectl-mcp-server/docker/) — - # NOT the upstream Docker Hub image. 266→18 HIGH/CRIT vulns, non-root, - # multi-stage slim base, kubectl v1.33.12 / helm v3.21.0. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/prd/devop/kubectl-mcp-server - tag: "v2" - pullPolicy: IfNotPresent - -labels: - bu: infra - team: devops - service: kubectl-mcp-server - env: prd - -serviceAccount: - create: true - annotations: {} - -rbac: - create: true - -podAnnotations: {} - -podSecurityContext: {} - -securityContext: - runAsNonRoot: true - runAsUser: 1000 - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -priorityClassName: "" - -nodeSelector: - dedicated: "supply-devops" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "supply-devops" - effect: NoSchedule - -affinity: {} - -resources: - requests: - cpu: 250m - memory: 512Mi - limits: - cpu: 1000m - memory: 1Gi - -mcp: - mode: single - transport: http - host: "0.0.0.0" - port: 8000 - -auth: - allowAnonymous: false - -externalSecrets: - enabled: true - refreshInterval: "150s" - secretStoreRef: - name: vault-backend - kind: ClusterSecretStore - dataFrom: - secretKey: "meesho/prd/cntr/devop/kubectl-mcp-server-supply" - -# tcpSocket on purpose — streamable-http transport exposes only /mcp, -# there is no /health route (see chart values.yaml note). -livenessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 3 - -readinessProbe: - tcpSocket: - port: 8000 - initialDelaySeconds: 10 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 3 - -service: - port: 8000 - type: ClusterIP - -# Contour standard (2.0.0-style): renders HTTPProxy parent/child + -intra -# instead of a networking.k8s.io Ingress. -createContourGateway: true - -ingress: - enabled: true - ingressClassName: contour-internal-1 - servicePortNumber: 8000 - hosts: - - host: kubectl-mcp-server-supply.prd.meesho.int - paths: - - path: / - pathType: Prefix - annotations: {} - slowStart: - enabled: false - window: "120s" - aggression: 1 - minPercent: 10 diff --git a/helm-overrides/k8s-supply-prd-ase1/kubernetes-dashboard/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/kubernetes-dashboard/custom-values.yaml deleted file mode 100644 index 670d841..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/kubernetes-dashboard/custom-values.yaml +++ /dev/null @@ -1,442 +0,0 @@ -# Copyright 2017 The Kubernetes Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -# General configuration shared across resources -app: - # Mode determines if chart should deploy a full Dashboard with all containers or just the API. - # - dashboard - deploys all the containers - # - api - deploys just the API - mode: 'dashboard' - image: - pullPolicy: IfNotPresent - pullSecrets: [] - scheduling: - # Node labels for pod assignment - # Ref: https://kubernetes.io/docs/user-guide/node-selection/ - nodeSelector: {} - security: - # Allow overriding csrfKey used by API/Auth containers. - # It has to be base64 encoded random 256 bytes string. - # If empty, it will be autogenerated. - csrfKey: ~ - # SecurityContext to be added to pods - # To disable set the following configuration to null: - # securityContext: null - securityContext: - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - # ContainerSecurityContext to be added to containers - # To disable set the following configuration to null: - # containerSecurityContext: null - containerSecurityContext: - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - runAsUser: 1001 - runAsGroup: 2001 - capabilities: - drop: ["ALL"] - # Pod Disruption Budget configuration - # Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - minAvailable: 0 - maxUnavailable: 0 - networkPolicy: - enabled: false - ingressDenyAll: false - # Raw network policy spec that overrides predefined spec - # Example: - # spec: - # egress: - # - ports: - # - port: 123 - spec: {} - - # Common labels & annotations shared across all deployed resources - labels: {} - annotations: {} - # Common priority class used for all deployed resources - priorityClassName: null - settings: - ## Global dashboard settings - global: - # # Cluster name that appears in the browser window title if it is set - # clusterName: "" - # # Max number of items that can be displayed on each list page - # itemsPerPage: 10 - # # Max number of labels that are displayed by default on most views. - # labelsLimit: 3 - # # Number of seconds between every auto-refresh of logs - # logsAutoRefreshTimeInterval: 5 - # # Number of seconds between every auto-refresh of every resource. Set 0 to disable - # resourceAutoRefreshTimeInterval: 10 - # # Hide all access denied warnings in the notification panel - # disableAccessDeniedNotifications: false - # # Hide all namespaces option in namespace selection dropdown to avoid accidental selection in large clusters thus preventing OOM errors - # hideAllNamespaces: false - # # Namespace that should be selected by default after logging in. - # defaultNamespace: default - # # List of namespaces that should be presented to user without namespace list privileges. - # namespaceFallbackList: - # - default - ## Pinned resources that will be displayed in dashboard's menu - pinnedResources: [] - # - kind: customresourcedefinition - # # Fully qualified name of a CRD - # name: prometheus.monitoring.coreos.com - # # Display name - # displayName: Prometheus - # # Is this CRD namespaced? - # namespaced: true - ingress: - enabled: false - hosts: - # Keep 'localhost' host only if you want to access Dashboard using 'kubectl port-forward ...' on: - # https://localhost:8443 - - localhost - # - kubernetes.dashboard.domain.com - ingressClassName: internal-nginx - # Use only if your ingress controllers support default ingress classes. - # If set to true ingressClassName will be ignored and not added to the Ingress resources. - # It should fall back to using IngressClass marked as the default. - useDefaultIngressClass: false - # This will append our Ingress with annotations required by our default configuration. - # nginx.ingress.kubernetes.io/backend-protocol: "HTTPS" - # nginx.ingress.kubernetes.io/ssl-passthrough: "true" - # nginx.ingress.kubernetes.io/ssl-redirect: "true" - useDefaultAnnotations: true - pathType: ImplementationSpecific - # If path is not the default (/), rewrite-target annotation will be added to the Ingress. - # It allows serving Kubernetes Dashboard on a sub-path. Make sure that the configured path - # does not conflict with gateway route configuration. - path: / - issuer: - name: selfsigned - # Scope determines what kind of issuer annotation will be used on ingress resource - # - default - adds 'cert-manager.io/issuer' - # - cluster - adds 'cert-manager.io/cluster-issuer' - # - disabled - disables cert-manager annotations - scope: default - tls: - enabled: true - # If provided it will override autogenerated secret name - secretName: "" - labels: {} - annotations: {} - # Use the following toleration if Dashboard can be deployed on a tainted control-plane nodes - # - key: node-role.kubernetes.io/control-plane - # effect: NoSchedule - tolerations: [] - affinity: {} - -auth: - role: auth - image: - repository: docker.io/kubernetesui/dashboard-auth - tag: 1.2.2 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: auth - containerPort: 8000 - protocol: TCP - args: [] - env: [] - volumeMounts: - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for Auth related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -# API deployment configuration -api: - role: api - image: - repository: docker.io/kubernetesui/dashboard-api - tag: 1.10.1 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: api - containerPort: 8000 - protocol: TCP - # Additional container arguments - # Full list of arguments: https://github.com/kubernetes/dashboard/blob/master/docs/common/arguments.md - # args: - # - --system-banner="Welcome to the Kubernetes Dashboard" - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store exec logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for API related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -# WEB UI deployment configuration -web: - role: web - image: - repository: docker.io/kubernetesui/dashboard-web - tag: 1.6.0 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - name: web - containerPort: 8000 - protocol: TCP - # Additional container arguments - # Full list of arguments: https://github.com/kubernetes/dashboard/blob/master/docs/common/arguments.md - # args: - # - --system-banner="Welcome to the Kubernetes Dashboard" - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - # Create on-disk volume to store exec logs (required) - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for WEB UI related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -### Metrics Scraper -### Container to scrape, store, and retrieve a window of time from the Metrics Server. -### refs: https://github.com/kubernetes/dashboard/tree/master/modules/metrics-scraper -metricsScraper: - enabled: true - role: metrics-scraper - image: - repository: docker.io/kubernetesui/dashboard-metrics-scraper - tag: 1.2.1 - scaling: - replicas: 1 - revisionHistoryLimit: 10 - containers: - ports: - - containerPort: 8000 - protocol: TCP - args: [] - # Additional container environment variables - # env: - # - name: SOME_VAR - # value: 'some value' - env: [] - # Additional volume mounts - # - mountPath: /kubeconfig - # name: dashboard-kubeconfig - # readOnly: true - volumeMounts: - # Create volume mount to store logs (required) - - mountPath: /tmp - name: tmp-volume - # TODO: Validate configuration - resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 250m - memory: 400Mi - livenessProbe: - httpGet: - scheme: HTTP - path: / - port: 8000 - initialDelaySeconds: 30 - timeoutSeconds: 30 - automountServiceAccountToken: true - # Additional volumes - # - name: dashboard-kubeconfig - # secret: - # defaultMode: 420 - # secretName: dashboard-kubeconfig - volumes: - - name: tmp-volume - emptyDir: {} - nodeSelector: {} - # Labels & annotations for Metrics Scraper related resources - labels: {} - annotations: {} - serviceLabels: {} - serviceAnnotations: {} - -## Optional Metrics Server sub-chart configuration -## Enable this if you don't already have metrics-server enabled on your cluster and -## want to use it with dashboard metrics-scraper -## refs: -## - https://github.com/kubernetes-sigs/metrics-server -## - https://github.com/kubernetes-sigs/metrics-server/tree/master/charts/metrics-server -metrics-server: - enabled: false - args: - - --kubelet-preferred-address-types=InternalIP - - --kubelet-insecure-tls - -## Required Kong sub-chart with DBless configuration to act as a gateway -## for our all containers. -kong: - enabled: true - admin: - tls: - enabled: false - ## Configuration reference: https://docs.konghq.com/gateway/3.6.x/reference/configuration - env: - dns_order: LAST,A,CNAME,AAAA,SRV - plugins: 'off' - nginx_worker_processes: 1 - ingressController: - enabled: false - manager: - enabled: false - dblessConfig: - configMap: kong-dbless-config - proxy: - type: ClusterIP - http: - enabled: true - -## Optional Cert Manager sub-chart configuration -## Enable this if you don't already have cert-manager enabled on your cluster. -cert-manager: - enabled: false - installCRDs: true - -## Optional Nginx Ingress sub-chart configuration -## Enable this if you don't already have nginx-ingress enabled on your cluster. -nginx: - enabled: false - controller: - electionID: ingress-controller-leader - ingressClassResource: - name: internal-nginx - default: false - controllerValue: k8s.io/internal-ingress-nginx - service: - type: ClusterIP - -## Extra configurations: -## - manifests -## - predefined roles -## - prometheus -## - etc... -extras: - # Extra Kubernetes manifests to be deployed - # manifests: - # - apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: additional-configmap - # data: - # mykey: myvalue - manifests: [] - serviceMonitor: - # Whether to create a Prometheus Operator service monitor. - enabled: false - # Here labels can be added to the serviceMonitor - labels: {} - # Here annotations can be added to the serviceMonitor - annotations: {} - # metrics.serviceMonitor.metricRelabelings Specify Metric Relabelings to add to the scrape endpoint - # ref: https://github.com/coreos/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # metrics.serviceMonitor.relabelings [array] Prometheus relabeling rules - relabelings: [] - # ServiceMonitor connection scheme. Defaults to HTTPS. - scheme: https - # ServiceMonitor connection tlsConfig. Defaults to {insecureSkipVerify:true}. - tlsConfig: - insecureSkipVerify: true diff --git a/helm-overrides/k8s-supply-prd-ase1/kyverno/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/kyverno/custom-values.yaml deleted file mode 100644 index f00cb4e..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/kyverno/custom-values.yaml +++ /dev/null @@ -1,2240 +0,0 @@ -global: - - # -- Internal settings used with `helm template` to generate install manifest - # @ignored - templating: - enabled: false - debug: false - version: ~ - - image: - # -- (string) Global value that allows to set a single image registry across all deployments. - # When set, it will override any values set under `.image.registry` across the chart. - registry: ~ - # -- (list) Global list of Image pull secrets - # When set, it will override any values set under `imagePullSecrets` under different components across the chart. - imagePullSecrets: [] - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - caCertificates: - # -- Global CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - # Individual controller values will override this global value - data: ~ - - # -- Global value to set single volume to be mounted for CA certificates for all deployments. - # Not used when `.Values.global.caCertificates.data` is defined - # Individual controller values will override this global value - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Additional container environment variables to apply to all containers and init containers - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Global node labels for pod assignment. Non-global values will override the global value. - nodeSelector: - dedicated: supply-kyverno - - # -- Global List of node taints to tolerate. Non-global values will override the global value. - tolerations: - - key: "dedicated" - operator: "Equal" - value: "supply-kyverno" - effect: "NoSchedule" - -# -- (string) Override the name of the chart -nameOverride: ~ - -# -- (string) Override the expanded name of the chart -fullnameOverride: ~ - -# -- (string) Override the namespace the chart deploys to -namespaceOverride: ~ - -upgrade: - # -- Upgrading from v2 to v3 is not allowed by default, set this to true once changes have been reviewed. - fromV2: false - -apiVersionOverride: - # -- (string) Override api version used to create `PodDisruptionBudget`` resources. - # When not specified the chart will check if `policy/v1/PodDisruptionBudget` is available to - # determine the api version automatically. - podDisruptionBudget: ~ - -rbac: - roles: - # -- Aggregate ClusterRoles to Kubernetes default user-facing roles. For more information, see [User-facing roles](https://kubernetes.io/docs/reference/access-authn-authz/rbac/#user-facing-roles) - aggregate: - admin: true - view: true - -# Use openreports.io as the API group for reporting -openreports: - # -- Enable OpenReports feature in controllers - enabled: false - # -- Whether to install CRDs from the upstream OpenReports chart. Setting this to true requires enabled to also be true. - installCrds: false - -# CRDs configuration -crds: - - # -- Whether to have Helm install the Kyverno CRDs, if the CRDs are not installed by Helm, they must be added before policies can be created - install: true - - reportsServer: - # -- Kyverno reports-server is used in your cluster - enabled: false - - groups: - - # -- Install CRDs in group `kyverno.io` - kyverno: - cleanuppolicies: true - clustercleanuppolicies: true - clusterpolicies: true - globalcontextentries: true - policies: true - policyexceptions: true - updaterequests: true - - # -- Install CRDs in group `policies.kyverno.io` - policies: - validatingpolicies: true - policyexceptions: true - imagevalidatingpolicies: true - namespacedimagevalidatingpolicies: true - mutatingpolicies: true - generatingpolicies: true - deletingpolicies: true - namespaceddeletingpolicies: true - namespacedvalidatingpolicies: true - - # -- Install CRDs in group `reports.kyverno.io` - reports: - clusterephemeralreports: true - ephemeralreports: true - - # -- Install CRDs in group `wgpolicyk8s.io` - wgpolicyk8s: - clusterpolicyreports: false ##lk - policyreports: false ##lk - - # -- Additional CRDs annotations (Replace=true avoids 262144-byte annotation limit on large CRDs) - annotations: - argocd.argoproj.io/sync-options: Replace=true - # strategy.spinnaker.io/replace: 'true' - - # -- Additional CRDs labels - customLabels: {} - - migration: - - # -- Enable CRDs migration using helm post upgrade hook - enabled: true - - # -- Resources to migrate - resources: - - cleanuppolicies.kyverno.io - - clustercleanuppolicies.kyverno.io - - clusterpolicies.kyverno.io - - globalcontextentries.kyverno.io - - policies.kyverno.io - - policyexceptions.kyverno.io - - updaterequests.kyverno.io - - deletingpolicies.policies.kyverno.io - - generatingpolicies.policies.kyverno.io - - imagevalidatingpolicies.policies.kyverno.io - - namespacedimagevalidatingpolicies.policies.kyverno.io - - mutatingpolicies.policies.kyverno.io - - namespaceddeletingpolicies.policies.kyverno.io - - namespacedvalidatingpolicies.policies.kyverno.io - - policyexceptions.policies.kyverno.io - - validatingpolicies.policies.kyverno.io - - image: - # -- (string) Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- (string) Image repository - repository: kyverno/kyverno-cli - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- (string) Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podResources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -# Configuration -config: - - # -- Create the configmap. - create: true - - # -- Preserve the configmap settings during upgrade. - preserve: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - # -- Enable registry mutation for container images. Enabled by default. - enableDefaultRegistryMutation: true - - # -- The registry hostname used for the image mutation. - defaultRegistry: docker.io - - # -- Exclude groups - excludeGroups: - - system:nodes - - # -- Exclude usernames - excludeUsernames: [] - # - '!system:kube-scheduler' - - # -- Exclude roles - excludeRoles: [] - - # -- Exclude roles - excludeClusterRoles: [] - - # -- Generate success events. - generateSuccessEvents: false - - # -- Resource types to be skipped by the Kyverno policy engine. - # Make sure to surround each entry in quotes so that it doesn't get parsed as a nested YAML list. - # These are joined together without spaces, run through `tpl`, and the result is set in the config map. - # @default -- See [values.yaml](values.yaml) - resourceFilters: - - '[Event,*,*]' - # - '[*/*,kube-system,*]' - - '[*/*,kube-public,*]' - - '[*/*,kube-node-lease,*]' - - '[Node,*,*]' - - '[Node/?*,*,*]' - - '[APIService,*,*]' - - '[APIService/?*,*,*]' - - '[TokenReview,*,*]' - - '[SubjectAccessReview,*,*]' - - '[SelfSubjectAccessReview,*,*]' - - '[Binding,*,*]' - - '[Pod/binding,*,*]' - - '[ReplicaSet,*,*]' - - '[ReplicaSet/?*,*,*]' - - '[EphemeralReport,*,*]' - - '[ClusterEphemeralReport,*,*]' - # exclude resources from the chart - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.admission-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.background-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.cleanup-controller.roleName" . }}:additional]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:core]' - - '[ClusterRole,*,{{ template "kyverno.reports-controller.roleName" . }}:additional]' - - '[ClusterRoleBinding,*,{{ template "kyverno.admission-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.background-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[ClusterRoleBinding,*,{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.serviceAccountName" . }}]' - - '[ServiceAccount,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[ServiceAccount/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.serviceAccountName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[Role,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.roleName" . }}]' - - '[RoleBinding,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.roleName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.configMapName" . }}]' - - '[ConfigMap,{{ include "kyverno.namespace" . }},{{ template "kyverno.config.metricsConfigMapName" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Deployment,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Deployment/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-*]' - - '[Pod,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Pod/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-*]' - - '[Job,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[Job/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.fullname" . }}-hook-pre-delete]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[NetworkPolicy,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[NetworkPolicy/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[PodDisruptionBudget,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[PodDisruptionBudget/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.background-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}-metrics]' - - '[Service,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[Service/?*,{{ include "kyverno.namespace" . }},{{ template "kyverno.reports-controller.name" . }}-metrics]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.admission-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.background-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.cleanup-controller.name" . }}]' - - '[ServiceMonitor,{{ if .Values.admissionController.serviceMonitor.namespace }}{{ .Values.admissionController.serviceMonitor.namespace }}{{ else }}{{ template "kyverno.namespace" . }}{{ end }},{{ template "kyverno.reports-controller.name" . }}]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.admission-controller.serviceName" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - '[Secret,{{ include "kyverno.namespace" . }},{{ template "kyverno.cleanup-controller.name" . }}.{{ template "kyverno.namespace" . }}.svc.*]' - - # -- Sets the threshold for the total number of UpdateRequests generated for mutateExisitng and generate policies. - updateRequestThreshold: 1000 - - # -- Defines the `namespaceSelector`/`objectSelector` in the webhook configurations. - # The Kyverno namespace is excluded if `excludeKyvernoNamespace` is `true` (default) - webhooks: - # Exclude namespaces - # namespaceSelector: lk-s2 - # matchExpressions: - # - key: kubernetes.io/metadata.name - # operator: NotIn - # values: - # - kube-systemm lk-s2 - # Exclude objects - # objectSelector: - # matchExpressions: - # - key: webhooks.kyverno.io/exclude - # operator: DoesNotExist - - # -- Defines annotations to set on webhook configurations. - webhookAnnotations: - # Example to disable admission enforcer on AKS: - # 'admissions.enforcer/disabled': 'true' lk-s3 - - # -- Defines labels to set on webhook configurations. - webhookLabels: {} - # Example to adopt webhook resources in ArgoCD: - # 'argocd.argoproj.io/instance': 'kyverno' - - # -- Defines match conditions to set on webhook configurations (requires Kubernetes 1.27+). - matchConditions: [] - - # -- Exclude Kyverno namespace - # Determines if default Kyverno namespace exclusion is enabled for webhooks and resourceFilters - excludeKyvernoNamespace: true - - # -- resourceFilter namespace exclude - # Namespaces to exclude from the default resourceFilters - resourceFiltersExcludeNamespaces: [] - - # -- resourceFilters exclude list - # Items to exclude from config.resourceFilters - resourceFiltersExclude: [] - - # -- resourceFilter namespace include - # Namespaces to include to the default resourceFilters - resourceFiltersIncludeNamespaces: [] - - # -- resourceFilters include list - # Items to include to config.resourceFilters - resourceFiltersInclude: [] - -# Metrics configuration -metricsConfig: - - # -- Create the configmap. - create: true - - # -- (string) The configmap name (required if `create` is `false`). - name: ~ - - # -- Additional annotations to add to the configmap. - annotations: {} - - namespaces: - - # -- List of namespaces to capture metrics for. - include: [] - - # -- list of namespaces to NOT capture metrics for. - exclude: [] - - # -- (string) Rate at which metrics should reset so as to clean up the memory footprint of kyverno metrics, if you might be expecting high memory footprint of Kyverno's metrics. Default: 0, no refresh of metrics. WARNING: This flag is not working since Kyverno 1.8.0 - metricsRefreshInterval: ~ - # metricsRefreshInterval: 24h - - # -- (list) Configures the bucket boundaries for all Histogram metrics, changing this configuration requires restart of the kyverno admission controller - bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 15, 20, 25, 30] - - # -- (map) Configures the exposure of individual metrics, by default all metrics and all labels are exported, changing this configuration requires restart of the kyverno admission controller - metricsExposure: - kyverno_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_image_validating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_mutating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_generating_policy_execution_duration_seconds: - # bucketBoundaries: [0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5] - disabledLabelDimensions: ["resource_namespace", "resource_request_operation"] - kyverno_admission_review_duration_seconds: - # enabled: false - disabledLabelDimensions: ["resource_namespace"] - kyverno_policy_rule_info_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_policy_results_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - kyverno_admission_requests_total: - disabledLabelDimensions: ["resource_namespace"] - kyverno_cleanup_controller_deletedobjects_total: - disabledLabelDimensions: ["resource_namespace", "policy_namespace"] - -# -- Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -imagePullSecrets: {} - # regcred: - # registry: foo.example.com - # username: foobar - # password: secret - # regcred2: - # registry: bar.example.com - # username: barbaz - # password: secret2 - -# -- Existing Image pull secrets for image verification policies, this will define the `--imagePullSecrets` argument -existingImagePullSecrets: [] - # - test-registry - # - other-test-registry - -# Tests configuration -test: - # -- Sleep time before running test - sleep: 20 - - image: - # -- (string) Image registry - registry: curlimages - # -- Image repository - repository: curl - # -- Image tag - # Defaults to `latest` if omitted - tag: '8.10.1' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - # - name: secretName - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - # -- Security context for the test containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- Additional Pod annotations - podAnnotations: {} - - # -- List of node taints to tolerate - tolerations: [] - -# -- Additional labels -customLabels: {} - - -webhooksCleanup: - # -- Create a helm pre-delete hook to cleanup webhooks. - enabled: true - - autoDeleteWebhooks: - # -- Allow webhooks controller to delete webhooks using finalizers - enabled: false - - image: - # -- (string) Image registry - registry: registry.k8s.io - # -- Image repository - repository: kubectl - # -- Image tag - # Defaults to `latest` if omitted - tag: 'v1.32.7' - # -- (string) Image pull policy - # Defaults to image.pullPolicy if omitted - pullPolicy: ~ - - # -- Image pull secrets - imagePullSecrets: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - # -- Pod anti affinity constraints. - podAntiAffinity: {} - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Pod labels. - podLabels: {} - - # -- Pod annotations. - podAnnotations: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Security context for the hook containers - securityContext: - runAsUser: 65534 - runAsGroup: 65534 - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - resources: - # -- Pod resource limits - limits: - cpu: 100m - memory: 256Mi - # -- Pod resource requests - requests: - cpu: 10m - memory: 64Mi - - serviceAccount: - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - -grafana: - # -- Enable grafana dashboard creation. - enabled: false - - # -- Configmap name template. - configMapName: '{{ include "kyverno.fullname" . }}-grafana' - - # -- (string) Namespace to create the grafana dashboard configmap. - # If not set, it will be created in the same namespace where the chart is deployed. - namespace: ~ - - # -- Grafana dashboard configmap annotations. - annotations: {} - - # -- Grafana dashboard configmap labels - labels: - grafana_dashboard: "1" - - # -- create GrafanaDashboard custom resource referencing to the configMap. - # according to https://grafana-operator.github.io/grafana-operator/docs/examples/dashboard_from_configmap/readme/ - grafanaDashboard: - create: false - folder: kyverno - allowCrossNamespaceImport: true - matchLabels: - dashboards: "grafana" - -# Features configuration -features: - admissionReports: - # -- Enables the feature - enabled: true - aggregateReports: - # -- Enables the feature - enabled: true - policyReports: - # -- Enables the feature - enabled: false ##lk - validatingAdmissionPolicyReports: - # -- Enables the feature - enabled: true - mutatingAdmissionPolicyReports: - # -- Enables the feature - enabled: false - reporting: - # -- Enables the feature - validate: true - # -- Enables the feature - mutate: true - # -- Enables the feature - mutateExisting: true - # -- Enables the feature - imageVerify: true - # -- Enables the feature - generate: true - autoUpdateWebhooks: - # -- Enables the feature - enabled: true - backgroundScan: - # -- Enables the feature - enabled: true - # -- Number of background scan workers - backgroundScanWorkers: 2 - # -- Background scan interval - backgroundScanInterval: 1h - # -- Skips resource filters in background scan - skipResourceFilters: true - configMapCaching: - # -- Enables the feature - enabled: true - controllerRuntimeMetrics: - # -- Bind address for controller-runtime metrics (use "0" to disable it) - bindAddress: ":8080" - deferredLoading: - # -- Enables the feature - enabled: true - dumpPayload: - # -- Enables the feature - enabled: false - forceFailurePolicyIgnore: - # -- Enables the feature - enabled: false - generateValidatingAdmissionPolicy: - # -- Enables the feature - enabled: true - generateMutatingAdmissionPolicy: - # -- Enables the feature - enabled: false - dumpPatches: - # -- Enables the feature - enabled: false - globalContext: - # -- Maximum allowed response size from API Calls. A value of 0 bypasses checks (not recommended) - maxApiCallResponseLength: 2000000 - logging: - # -- Logging format - format: text - # -- Logging verbosity - verbosity: 2 - omitEvents: - # -- Events which should not be emitted (possible values `PolicyViolation`, `PolicyApplied`, `PolicyError`, and `PolicySkipped`) - eventTypes: - - PolicyApplied - - PolicySkipped - # - PolicyViolation - # - PolicyError - policyExceptions: - # -- Enables the feature - enabled: true - # -- Restrict policy exceptions to a single namespace - # Set to "*" to allow exceptions in all namespaces - namespace: '' - protectManagedResources: - # -- Enables the feature - enabled: false - registryClient: - # -- Allow insecure registry - allowInsecure: false - # -- Enable registry client helpers - credentialHelpers: - - default - - google - - amazon - - azure - - github - ttlController: - # -- Reconciliation interval for the label based cleanup manager - reconciliationInterval: 1m - tuf: - # -- Enables the feature - enabled: false - # -- (string) Path to Tuf root - root: ~ - # -- (string) Raw Tuf root - rootRaw: ~ - # -- (string) Tuf mirror - mirror: ~ - -# Admission controller configuration -admissionController: - autoscaling: - # -- Enable horizontal pod autoscaling - enabled: true - - # -- Minimum number of pods - minReplicas: 1 - - # -- Maximum number of pods - maxReplicas: 10 - - # -- Target CPU utilization percentage - targetCPUUtilizationPercentage: 80 - - # -- Configurable scaling behavior - behavior: {} - - # -- Overrides features defined at the root level - featuresOverride: - admissionReports: - # -- Max number of admission reports allowed in flight until the admission controller stops creating new ones - backPressureThreshold: 1000 - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- The ServiceAccount name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Enable/Disable custom resource watcher to invalidate cache - crdWatcher: false - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno admission controller activities. - # This will help ensure Kyverno stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- admissionController webhook server port - # in case you are using hostNetwork: true, you might want to change the port the webhookServer is listening to - webhookServer: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - admission-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.admissionController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - initContainer: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyvernopre - # -- (string) Image tag - # If missing, defaults to image.tag - tag: ~ - # -- (string) Image pull policy - # If missing, defaults to image.pullPolicy - pullPolicy: ~ - - resources: - # -- Pod resource limits - # limits: - # cpu: 100m - # memory: 256Mi - # -- Pod resource requests - requests: - cpu: 1000m - memory: 1Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - container: - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/kyverno - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - resources: - # -- Pod resource limits - # limits: - # memory: 384Mi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Container security context - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - # -- Additional container args. - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - # -- Array of extra init containers - extraInitContainers: [] - # - name: init-container - # image: busybox - # command: ['sh', '-c', 'echo Hello'] - - # -- Array of extra containers to run alongside kyverno - extraContainers: [] - # - name: myapp-container - # image: busybox - # command: ['sh', '-c', 'echo Hello && sleep 3600'] - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Kyverno's metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Background controller configuration -backgroundController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable background controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: - - apiGroups: - - networking.k8s.io - resources: - - ingresses - - ingressclasses - - networkpolicies - verbs: - - create - - update - - patch - - delete - - apiGroups: - - rbac.authorization.k8s.io - resources: - - rolebindings - - roles - verbs: - - create - - update - - patch - - delete - - apiGroups: - - '' - resources: - - configmaps - - resourcequotas - - limitranges - verbs: - - create - - update - - patch - - delete - - apiGroups: - - resource.k8s.io - resources: - - resourceclaims - - resourceclaimtemplates - verbs: - - create - - delete - - update - - patch - - deletecollection - - apiGroups: - - autoscaling - resources: - - horizontalpodautoscalers - verbs: - - get - - list - - watch - - update - - patch - - apiGroups: - - apps - resources: - - deployments - - statefulsets - - replicasets - verbs: - - get - - list - - watch - - update - - patch - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - create - # - update - # - delete - # - patch - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/background-controller - # -- Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - # limits: - # memory: 128Mi - # -- Pod resource requests - requests: - cpu: 1900m - memory: 2Gi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - background-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: true - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.backgroundController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - # -- backgroundController server port - # in case you are using hostNetwork: true, you might want to change the port the backgroundController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Cleanup controller configuration -cleanupController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable cleanup controller. - enabled: true - - rbac: - # -- Create RBAC resources - create: true - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - # verbs: - # - delete - # - list - # - watch - - # -- Create self-signed certificates at deployment time. - # The certificates won't be automatically renewed if this is set to `true`. - createSelfSignedCert: false - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/cleanup-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- cleanupController server port - # in case you are using hostNetwork: true, you might want to change the port the cleanupController is listening to - server: - port: 9443 - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - # limits: - # memory: 128Mi - # -- Pod resource requests - requests: - cpu: 1100m - memory: 1.2Gi - - # -- Startup probe. - # The block is directly forwarded into the deployment, so you can use whatever startupProbes configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - startupProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - failureThreshold: 20 - initialDelaySeconds: 2 - periodSeconds: 6 - - # -- Liveness probe. - # The block is directly forwarded into the deployment, so you can use whatever livenessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - livenessProbe: - httpGet: - path: /health/liveness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 15 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 2 - successThreshold: 1 - - # -- Readiness Probe. - # The block is directly forwarded into the deployment, so you can use whatever readinessProbe configuration you want. - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-probes/ - # @default -- See [values.yaml](values.yaml) - readinessProbe: - httpGet: - path: /health/readiness - port: 9443 - scheme: HTTPS - initialDelaySeconds: 5 - periodSeconds: 10 - timeoutSeconds: 5 - failureThreshold: 6 - successThreshold: 1 - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - cleanup-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - service: - # -- Service port. - port: 443 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `service.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- Service node port. - # Only used if `metricsService.type` is `NodePort`. - nodePort: - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- Traces receiver address - address: - # -- Traces receiver port - port: - # -- Traces receiver credentials - creds: '' - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- Otel collector endpoint - collector: '' - # -- Otel collector credentials - creds: '' - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - -# Reports controller configuration -reportsController: - - # -- Overrides features defined at the root level - featuresOverride: {} - - # -- Enable reports controller. - enabled: false ##lk - - rbac: - # -- Create RBAC resources - create: true - - # -- Create rolebinding to view role - createViewRoleBinding: true - - # -- The view role to use in the rolebinding - viewRoleName: view - - serviceAccount: - # -- Service account name - name: - - # -- Annotations for the ServiceAccount - annotations: {} - # example.com/annotation: value - - # -- Toggle automounting of the ServiceAccount - automountServiceAccountToken: true - - coreClusterRole: - # -- Extra resource permissions to add in the core cluster role. - # This was introduced to avoid breaking change in the chart but should ideally be moved in `clusterRole.extraResources`. - # @default -- See [values.yaml](values.yaml) - extraResources: [] - - clusterRole: - # -- Extra resource permissions to add in the cluster role - extraResources: [] - # - apiGroups: - # - '' - # resources: - # - pods - - image: - # -- Image registry - registry: ~ - defaultRegistry: reg.kyverno.io - # -- Image repository - repository: kyverno/reports-controller - # -- (string) Image tag - # Defaults to appVersion in Chart.yaml if omitted - tag: ~ - # -- Image pull policy - pullPolicy: IfNotPresent - - # -- Image pull secrets - imagePullSecrets: [] - # - secretName - - # -- (int) Desired number of pods - replicas: ~ - - # -- The number of revisions to keep - revisionHistoryLimit: 10 - - # -- Resync period for informers - resyncPeriod: 15m - - # -- Additional labels to add to each pod - podLabels: {} - # example.com/label: foo - - # -- Additional annotations to add to each pod - podAnnotations: {} - # example.com/annotation: foo - - # -- Deployment annotations. - annotations: {} - - # -- Deployment update strategy. - # Ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - # @default -- See [values.yaml](values.yaml) - updateStrategy: - rollingUpdate: - maxSurge: 1 - maxUnavailable: 40% - type: RollingUpdate - - # -- Optional priority class - priorityClassName: '' - - # -- Change `apiPriorityAndFairness` to `true` if you want to insulate the API calls made by Kyverno reports controller activities. - # This will help ensure Kyverno reports stability in busy clusters. - # Ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/ - apiPriorityAndFairness: false - - # -- Priority level configuration. - # The block is directly forwarded into the priorityLevelConfiguration, so you can use whatever specification you want. - # ref: https://kubernetes.io/docs/concepts/cluster-administration/flow-control/#prioritylevelconfiguration - # @default -- See [values.yaml](values.yaml) - priorityLevelConfigurationSpec: - type: Limited - limited: - nominalConcurrencyShares: 10 - limitResponse: - queuing: - queueLengthLimit: 50 - type: Queue - - # -- Change `hostNetwork` to `true` when you want the pod to share its host's network namespace. - # Useful for situations like when you end up dealing with a custom CNI over Amazon EKS. - # Update the `dnsPolicy` accordingly as well to suit the host network mode. - hostNetwork: false - - # -- `dnsPolicy` determines the manner in which DNS resolution happens in the cluster. - # In case of `hostNetwork: true`, usually, the `dnsPolicy` is suitable to be `ClusterFirstWithHostNet`. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-s-dns-policy. - dnsPolicy: ClusterFirst - - # -- `dnsConfig` allows to specify DNS configuration for the pod. - # For further reference: https://kubernetes.io/docs/concepts/services-networking/dns-pod-service/#pod-dns-config. - dnsConfig: {} - # options: - # - name: ndots - # value: "2" - - # -- Extra arguments passed to the container on the command line - extraArgs: {} - - # -- Additional container environment variables. - extraEnvVars: [] - # Example setting proxy - # extraEnvVars: - # - name: HTTPS_PROXY - # value: 'https://proxy.example.com:3128' - - resources: - # -- Pod resource limits - limits: - memory: 128Mi - # -- Pod resource requests - requests: - cpu: 100m - memory: 64Mi - - # -- Node labels for pod assignment - nodeSelector: {} - - # -- List of node taints to tolerate - tolerations: [] - - antiAffinity: - # -- Pod antiAffinities toggle. - # Enabled by default but can be disabled if you want to schedule pods to the same node. - enabled: true - - # -- Pod anti affinity constraints. - # @default -- See [values.yaml](values.yaml) - podAntiAffinity: - preferredDuringSchedulingIgnoredDuringExecution: - - weight: 1 - podAffinityTerm: - labelSelector: - matchExpressions: - - key: app.kubernetes.io/component - operator: In - values: - - reports-controller - topologyKey: kubernetes.io/hostname - - # -- Pod affinity constraints. - podAffinity: {} - - # -- Node affinity constraints. - nodeAffinity: {} - - # -- Topology spread constraints. - topologySpreadConstraints: [] - - # -- Security context for the pod - podSecurityContext: {} - - # -- Security context for the containers - securityContext: - runAsNonRoot: true - privileged: false - allowPrivilegeEscalation: false - readOnlyRootFilesystem: true - capabilities: - drop: - - ALL - seccompProfile: - type: RuntimeDefault - - podDisruptionBudget: - # -- Enable PodDisruptionBudget. - # Will always be enabled if replicas > 1. This non-declarative behavior should ideally be avoided, but changing it now would be breaking. - enabled: false - # -- Configures the minimum available pods for disruptions. - # Cannot be used if `maxUnavailable` is set. - minAvailable: 1 - # -- Configures the maximum unavailable pods for disruptions. - # Cannot be used if `minAvailable` is set. - maxUnavailable: - # -- Unhealthy pod eviction policy to be used. - # Possible values are `IfHealthyBudget` or `AlwaysAllow`. - unhealthyPodEvictionPolicy: - - # -- A writable volume to use for the TUF root initialization. - tufRootMountPath: /.sigstore - - # -- Volume to be mounted in pods for TUF/cosign work. - sigstoreVolume: - emptyDir: {} - - caCertificates: - # -- CA certificates to use with Kyverno deployments - # This value is expected to be one large string of CA certificates - data: ~ - # -- Volume to be mounted for CA certificates - # Not used when `.Values.reportsController.caCertificates.data` is defined - volume: {} - # Example to use hostPath: - # hostPath: - # path: /etc/pki/tls/ca-certificates.crt - # type: File - - - metricsService: - # -- Create service. - create: true - # -- Service port. - # Metrics server will be exposed at this port. - port: 8000 - # -- Service type. - type: ClusterIP - # -- (string) Service node port. - # Only used if `type` is `NodePort`. - nodePort: ~ - # -- Service annotations. - annotations: {} - # -- (string) Service traffic distribution policy. - # Set to `PreferClose` to route traffic to nearby endpoints, reducing latency and cross-zone costs. - trafficDistribution: ~ - - networkPolicy: - - # -- When true, use a NetworkPolicy to allow ingress to the webhook - # This is useful on clusters using Calico and/or native k8s network policies in a default-deny setup. - enabled: false - - # -- A list of valid from selectors according to https://kubernetes.io/docs/concepts/services-networking/network-policies. - ingressFrom: [] - - serviceMonitor: - # -- Create a `ServiceMonitor` to collect Prometheus metrics. - enabled: false - # -- Additional annotations - additionalAnnotations: {} - # -- Additional labels - additionalLabels: {} - # -- (string) Override namespace - namespace: ~ - # -- Interval to scrape metrics - interval: 30s - # -- Timeout if metrics can't be retrieved in given time interval - scrapeTimeout: 25s - # -- Is TLS required for endpoint - secure: false - # -- TLS Configuration for endpoint - tlsConfig: {} - # -- RelabelConfigs to apply to samples before scraping - relabelings: [] - # -- MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - tracing: - # -- Enable tracing - enabled: false - # -- (string) Traces receiver address - address: ~ - # -- (string) Traces receiver port - port: ~ - # -- (string) Traces receiver credentials - creds: ~ - - metering: - # -- Disable metrics export - disabled: false - # -- Otel configuration, can be `prometheus` or `grpc` - config: prometheus - # -- Prometheus endpoint port - port: 8000 - # -- (string) Otel collector endpoint - collector: ~ - # -- (string) Otel collector credentials - creds: ~ - - # -- reportsController server port - # in case you are using hostNetwork: true, you might want to change the port the reportsController is listening to - server: - port: 9443 - - profiling: - # -- Enable profiling - enabled: false - # -- Profiling endpoint port - port: 6060 - # -- Service type. - serviceType: ClusterIP - # -- Service node port. - # Only used if `type` is `NodePort`. - nodePort: - - # -- Enable sanity check for reports CRDs - sanityChecks: true diff --git a/helm-overrides/k8s-supply-prd-ase1/kyverno/policies/protect-namespace.yaml b/helm-overrides/k8s-supply-prd-ase1/kyverno/policies/protect-namespace.yaml deleted file mode 100644 index b07c26f..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/kyverno/policies/protect-namespace.yaml +++ /dev/null @@ -1,40 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: protect-critical-namespaces -spec: - validationFailureAction: Enforce - background: false - rules: - - name: deny-deletion-of-critical-namespaces - match: - any: - - resources: - kinds: - - Namespace - names: - - keda-supply-prd - - contour-internal-0-supply-prd - - contour-internal-0-supply-prd-intra - - contour-external-supply-prd - - external-secrets-supply-prd - - flagger-supply-prd - - victoriametrics - - kube-system - - monitoring - - telegraf-operator - - loadtester - - opentelemetry - - fluentd - - kube-events - - observability - - contour-cert-checker-ns - - aurva-dataplane - validate: - message: "Deletion of critical namespaces is not allowed." - deny: - conditions: - any: - - key: "{{request.operation}}" - operator: Equals - value: DELETE diff --git a/helm-overrides/k8s-supply-prd-ase1/kyverno/policies/restrict-replicas.yaml b/helm-overrides/k8s-supply-prd-ase1/kyverno/policies/restrict-replicas.yaml deleted file mode 100644 index d456985..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/kyverno/policies/restrict-replicas.yaml +++ /dev/null @@ -1,29 +0,0 @@ -apiVersion: kyverno.io/v1 -kind: ClusterPolicy -metadata: - name: restrict-minimum-replicas-selected-ns -spec: - validationFailureAction: Enforce - background: false - rules: - - name: min-replicas-deployments-statefulsets - match: - resources: - kinds: - - Deployment - - StatefulSet - - Deployment/scale - - StatefulSet/scale - names: - - coredns - namespaces: - - kube-system - operations: - - CREATE - - UPDATE - validate: - message: "Deployments and StatefulSets must have at least 5 replicas in this namespace." - anyPattern: - # Case 1: replicas explicitly set and >= 5 - - spec: - replicas: ">=5" diff --git a/helm-overrides/k8s-supply-prd-ase1/loadtester/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/loadtester/custom-values.yaml deleted file mode 100644 index a1620f7..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/loadtester/custom-values.yaml +++ /dev/null @@ -1,116 +0,0 @@ -replicaCount: 1 - -Namespace: loadtester - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/dev/devops/flagger-loadtester - tag: 0.30.0 - pullPolicy: IfNotPresent - pullSecret: - -podLabels: - bu: supply - team: supply-shared - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - -podPriorityClassName: "" - -logLevel: info -cmd: - timeout: 1h - namespaceRegexp: "" - -nameOverride: "loadtester" -fullnameOverride: "" - -env: [] - -service: - type: ClusterIP - port: 80 - -resources: - requests: - cpu: 10m - memory: 64Mi - -volumes: [] -volumeMounts: [] - -nodeSelector: - dedicated: supply-devops - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "supply-devops" - effect: "NoSchedule" - -affinity: {} - -ingress: - slowStart: - enabled: false - enabled: true - ingressClassName: contour-internal-1 - servicePort: http - hosts: - - host: supply-flagger-loadtester.prd.meesho.int - paths: - - pathType: ImplementationSpecific - path: / -createContourGateway: true -contourResponseTimeout: false -rbac: - # rbac.create: `true` if rbac resources should be created - create: true - # rbac.scope: `cluster` to create cluster-scope rbac resources (ClusterRole/ClusterRoleBinding) - # otherwise, namespace-scope rbac resources will be created (Role/RoleBinding) - scope: - # rbac.rules: array of rules to apply to the role. example: - # rules: - # - apiGroups: [""] - # resources: ["pods"] - # verbs: ["list", "get"] - rules: [] - -# name of an existing service account to use - if not creating rbac resources -serviceAccountName: "" - -# App Mesh virtual node settings (to be used for AppMesh v1beta1) -meshName: "" -#backends: -# - app1.namespace -# - app2.namespace - -# App Mesh virtual node settings (to be used for AppMesh v1beta2) -appmesh: - enabled: false - backends: - - podinfo - - podinfo-canary - -#Istio virtual service and gatway settings. TLS secrets should be in namespace before enbaled it. ( secret format loadtester.fullname ) -istio: - enabled: false - host: flagger-loadtester.flagger - gateway: - enabled: false - tls: - enabled: false - httpsRedirect: false - -# when enabled, it will add a security context for the loadtester pod -securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 100 - runAsGroup: 101 - -podDisruptionBudget: - enabled: false - minAvailable: 1 diff --git a/helm-overrides/k8s-supply-prd-ase1/node-thp-config/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/node-thp-config/custom-values.yaml deleted file mode 100644 index ec67538..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/node-thp-config/custom-values.yaml +++ /dev/null @@ -1,9 +0,0 @@ -daemonSet: - namespace: prd-node-thp-config - -baseMatchExpressions: -- key: dedicated - operator: In - values: - - "sumounolite-azul" # change to your actual node label value - - "megatetralite-azul" # adding megatetralite diff --git a/helm-overrides/k8s-supply-prd-ase1/opentelemetry-coralogix/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/opentelemetry-coralogix/custom-values.yaml deleted file mode 100644 index 08afac2..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/opentelemetry-coralogix/custom-values.yaml +++ /dev/null @@ -1,288 +0,0 @@ -global: - domain: "coralogixsg.com" - defaultApplicationName: "default" - defaultSubsystemName: "nodes" - - # Old endpoint based configuration, - # please use domain instead. - traces: - endpoint: "" - metrics: - endpoint: "" - logs: - endpoint: "" - -# set distribution to openshift for openshift clusters -distribution: "" - -# Opentelemetry collector configuration -dedicatedValue: false -schedulerName: default-scheduler - -labels: - bu: "supply" - team: "sre" - service: "opentelemetry-supply-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -externalSecret: - enabled: false - key: prd/admin/coralogix-keys - secretStoreRef: - name: vault-backend - -image: - # If you want to use the core image `otel/opentelemetry-collector`, you also need to change `command.name` value to `otelcol`. - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - # repository: otel/opentelemetry-collector-contrib - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - tag: "0.87.0" - # When digest is set to a non-empty value, images will be pulled by digest (regardless of tag value). - digest: "" -mode: daemonset -rollout: - rollingUpdate: - # When 'mode: daemonset', maxSurge cannot be used when hostPort is set for any of the ports - # maxSurge: 25% - maxUnavailable: 15 - strategy: RollingUpdate -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" -hostNetwork: true -dnsPolicy: "ClusterFirstWithHostNet" -fullnameOverride: opentelemetry-supply-prd - -presets: - logsCollection: - enabled: false - storeCheckpoints: false - maxRecombineLogSize: 1048576 - extraFilelogOperators: [] -# - type: recombine -# combine_field: body -# source_identifier: attributes["log.file.path"] -# is_first_entry: body matches "^(YOUR-LOGS-REGEX)" - kubernetesAttributes: - enabled: false - hostMetrics: - enabled: false - kubeletMetrics: - enabled: false - -extraEnvs: -- name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" -- name: KUBE_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName -config: - extensions: {} - # zpages: - # endpoint: localhost:55679 - # pprof: - # endpoint: localhost:1777 - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 2s - tls: - insecure: true - sending_queue: - enabled: true - queue_size: 30000 - num_consumers: 500 - resolver: - dns: - hostname: opentelemetry-admin-prd.opentelemetry.svc.clusterset.local - # coralogix: - # timeout: "30s" - # private_key: "${CORALOGIX_PRIVATE_KEY}" - # domain: "{{.Values.global.domain}}" - # traces: - # endpoint: "{{ .Values.global.traces.endpoint }}" - # metrics: - # endpoint: "{{ .Values.global.metrics.endpoint }}" - # logs: - # endpoint: "{{ .Values.global.logs.endpoint }}" - # application_name_attributes: - # - "k8s.namespace.name" - # - "service.namespace" - # subsystem_name_attributes: - # - "k8s.deployment.name" - # - "k8s.statefulset.name" - # - "k8s.daemonset.name" - # - "k8s.cronjob.name" - # - "k8s.job.name" - # - "k8s.container.name" - # - "k8s.node.name" - # - "service.name" - # application_name: "{{.Values.global.defaultApplicationName }}" - # subsystem_name: "{{.Values.global.defaultSubsystemName }}" - processors: - k8sattributes: - filter: - node_from_env_var: KUBE_NODE_NAME - extract: - metadata: - - "k8s.namespace.name" - - "k8s.deployment.name" - - "k8s.statefulset.name" - - "k8s.daemonset.name" - - "k8s.cronjob.name" - - "k8s.job.name" - - "k8s.pod.name" - - "k8s.node.name" - memory_limiter: null # Will get the k8s resource limits - groupbytrace: - wait_duration: 1s - groupbyattrs: - keys: - - host.name - resourcedetection/env: - detectors: ["system","env"] - timeout: 5s - override: false - # spanmetrics: - # metrics_exporter: coralogix - # dimensions: - # - name: "k8s.deployment.name" - # - name: "k8s.statefulset.name" - # - name: "k8s.daemonset.name" - # - name: "k8s.cronjob.name" - # - name: "k8s.job.name" - # - name: "k8s.container.name" - # - name: "k8s.node.name" - # - name: "k8s.namespace.name" - receivers: - otlp: - protocols: - grpc: - endpoint: ${MY_POD_IP}:4317 - http: - endpoint: ${MY_POD_IP}:4318 - zipkin: - endpoint: ${MY_POD_IP}:9411 - jaeger: - protocols: - grpc: - endpoint: ${MY_POD_IP}:14250 - thrift_http: - endpoint: ${MY_POD_IP}:14268 - thrift_compact: - endpoint: ${MY_POD_IP}:6831 - thrift_binary: - endpoint: ${MY_POD_IP}:6832 - prometheus: - config: - scrape_configs: - - job_name: opentelemetry-collector - scrape_interval: 30s - static_configs: - - targets: - - ${MY_POD_IP}:8888 - service: - extensions: - # - zpages - # - pprof - - health_check -# - memory_ballast - telemetry: - metrics: - address: ${env:MY_POD_IP}:8888 - pipelines: - traces: - exporters: - - loadbalancing - processors: - - memory_limiter - - batch - receivers: - - otlp - metrics: null - logs: null -tolerations: - - operator: Exists - -resources: - requests: - cpu: 100m - memory: 200Mi - limits: - cpu: 1 - memory: 2G - -ports: - jaeger-binary: - enabled: true - containerPort: 6832 - servicePort: 6832 - hostPort: 6832 - protocol: TCP - # In order to enable podMonitor, following part must be enabled in order to expose the required port: - # metrics: - # enabled: true - -# podMonitor: -# enabled: true - -# prometheusRule: -# enabled: true -# defaultRules: -# enabled: true - -# Annotations to be added to pod -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-supply-contour-ext - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-supply-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external diff --git a/helm-overrides/k8s-supply-prd-ase1/opentelemetry-daemonset-medium-np/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/opentelemetry-daemonset-medium-np/custom-values.yaml deleted file mode 100644 index cc8fe14..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/opentelemetry-daemonset-medium-np/custom-values.yaml +++ /dev/null @@ -1,110 +0,0 @@ -fullnameOverride: opentelemetry-medium-np-supply-prd - -mode: daemonset - -priorityClassName: "system-node-critical" - -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: In - values: - - sumoduolite-op - - key: dedicated - operator: In - values: - - sumounolite -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 350m - memory: 350Mi - limits: - cpu: 400m - memory: 512Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "supply" - team: "sre" - service: "opentelemetry-supply-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% diff --git a/helm-overrides/k8s-supply-prd-ase1/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index aef4c9d..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,135 +0,0 @@ -fullnameOverride: opentelemetry-supply-prd - -mode: daemonset - -priorityClassName: "system-node-critical" - -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-admin-prd-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: p-supply-contour-ext - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-0 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int-1 - operator: NotIn - values: - - dedicated - - key: p-supply-contour-int - operator: NotIn - values: - - dedicated - - key: node_pool - operator: NotIn - values: - - np-supply-default-prd-ase1 - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - contour-internal-0 - - contour-internal-1 - - contour-external - - alloy - - sumoduolite-op - - sumounolite -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 200m - memory: 256Mi - limits: - cpu: 400m - memory: 512Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "supply" - team: "sre" - service: "opentelemetry-supply-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% diff --git a/helm-overrides/k8s-supply-prd-ase1/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index a6ac9ce..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,496 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-supply-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -# Extra labels to be added to node exporter pods -podLabels: - bu: "supply" - team: "supply-sre" - service: "node-exporter-supply-prd" - env: "prd" - priority: "p0" - type: "exporter" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-supply-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index dfc2098..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-supply-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-supply-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "supply" - team: "supply-sre" - service: "stackdriver-exporter-supply-prd" - env: "prd" - priority: "p0" - type: "exporter" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-supply-prd-0622" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,compute.googleapis.com/instance,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - dedicated: "vmselect-mds" - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-supl-spsre-stackdriver-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - } - - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/k8s-supply-prd-ase1/telegraf-operator-custom/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/telegraf-operator-custom/custom-values.yaml deleted file mode 100644 index 3776013..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/telegraf-operator-custom/custom-values.yaml +++ /dev/null @@ -1,150 +0,0 @@ -replicaCount: 3 -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf-operator" - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "supply" - team: "supply-sre" - service: "telegraf-operator-custom-supply-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - dedicated: "vmselect-mds" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false diff --git a/helm-overrides/k8s-supply-prd-ase1/telegraf-operator/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/telegraf-operator/custom-values.yaml deleted file mode 100644 index 8dd8063..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,224 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - - infra-histogram-optimized: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 80000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - expiration_interval = "1m" - [[aggregators.histogram.config]] - buckets = [10.0, 25.0, 50.0, 100.0, 250.0, 500.0, 1000.0, 2500.0, 5000.0, 10000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "supply" - team: "supply-sre" - service: "telegraf-operator-supply-prd" - env: "prd" - priority: "p0" - type: "exporter" - -nodeSelector: - dedicated: "vmselect-mds" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false \ No newline at end of file diff --git a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml deleted file mode 100644 index 7359682..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-agent-dr/custom-values.yaml +++ /dev/null @@ -1,295 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-supply-prd-dr -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-supply-prd-dr-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-supl-spsre-vmagent-prd@meesho-supply-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - - https://vminsert-infra-prd-dr.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "supply" - team: "supply-sre" - service: "vmagent-supply-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-supply-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-supply-prd-dr.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 12 - memory: 12Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent-dr" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml deleted file mode 100644 index 664ffa9..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-agent-fb/custom-values.yaml +++ /dev/null @@ -1,272 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 1 - -fullnameOverride: vmagent-supply-prd-fb -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-supply-prd-fb-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - -# Remove topology spread constraints -topologySpreadConstraints: [] - -# Modify deployment strategy -deployment: - enabled: true - -# Remove HPA -horizontalPodAutoscaler: - enabled: false - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-infr-sre-vmagent-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -# Remove pod disruption budget -podDisruptionBudget: - enabled: false - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-supply.meeshogcp.in/insert/100/prometheus/api/v1/write - # - http://vmsinglenode-infra-prd.victoriametrics.svc.clusterset.local:8428/api/v1/write - - https://vmsinglenode-infra-prd.meeshogcp.in/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "supply-fb" - team: "sre" - service: "vmagent-supply-prd-fb" - env: "prd" - priority: "p0" - type: "vmagent-fb" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-supply-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-supply-prd-fb.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -# Modify resources -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - requests: - cpu: 2 - memory: 3Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-agent/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index eba8326..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,298 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-supply-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-supply-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-supl-sre-vmagnt-prd-mds@meesho-supply-prd-0622.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - https://vminsert-prd-supply.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-supply-prd.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write - - http://vminsert-ht-supply-prd.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write - - -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "supply" - team: "supply-sre" - service: "vmagent-supply-prd" - env: "prd" - priority: "p0" - type: "vmagent" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-supply-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-supply-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 55 - memory: 100Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent-mds" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-mds" - effect: "NoSchedule" - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml deleted file mode 100644 index 4feb7c9..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert-dr/custom-values.yaml +++ /dev/null @@ -1,304 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-supply-prd-dr - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 2 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-infra-prd-dr.victoriametrics.svc.clusterset.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-infra-prd-dr.victoriametrics.svc.clusterset.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/supply/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-supply-prd-dr.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-supply-prd-dr" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - podLabels: {} - - nodeSelector: - dedicated: "vmagent-dr" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-dr" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert-secured-stateful/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert-secured-stateful/custom-values.yaml deleted file mode 100644 index 25ffcd2..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert-secured-stateful/custom-values.yaml +++ /dev/null @@ -1,332 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-stateful-secured-supply-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-supply-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-supply-prd.victoriametrics.svc.cluster.local:8480/insert/200/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-supply-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/sensitive/supply/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: [] - # Extra Volume Mounts for the container - extraVolumeMounts: [] - extraContainers: [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: false - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-supply-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "supply" - team: "sre" - service: "vmalert-stateful-secured-supply-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: "false" - - podLabels: {} - - nodeSelector: - dedicated: "vmselect-mds" - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-secured-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert-secured/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert-secured/custom-values.yaml deleted file mode 100644 index 868e184..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert-secured/custom-values.yaml +++ /dev/null @@ -1,305 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-secured-supply-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-supply-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-supply-prd.victoriametrics.svc.cluster.local:8480/insert/200/prometheus" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-supply-prd-proxy.victoriametrics.svc.cluster.local:8481/select/200/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.cluster.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config1/sensitive/supply/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-secured-supply-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 1 - memory: 2Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "supply" - team: "sre" - service: "vmalert-secured-supply-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml deleted file mode 100644 index 2695870..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert-stateful/custom-values.yaml +++ /dev/null @@ -1,340 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-supply-prd-stateful - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-supply-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vm-insert-supply-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-supply-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/alert-rules/vm-alerts-config/configmap/supply/**/*.yaml" - evaluationInterval: 60s - configCheckInterval: 10s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: null # Set to null to prevent default values.yaml labels from being merged - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-supply-prd-stateful.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 1500m - memory: 3Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "supply" - team: "sre" - service: "vmalert-supply-prd-stateful" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - # -- Persistent Volume configuration for alert rules - persistentVolume: - # -- Create/use Persistent Volume Claim for alert rules. Empty dir if false - enabled: true - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteMany - - # -- Persistent volume annotations - annotations: {} - - # -- PVC extra labels - extraLabels: {} - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "pulse-nfs-sc-prd" - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Mount path. Alert rules Persistent Volume mount root path. - mountPath: /alert-rules - - # -- Mount subpath - subPath: "" - - # -- Size of the volume. Better to set the same as resource limit memory property. - size: 10Gi - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert/custom-values.yaml deleted file mode 100644 index 86d220f..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-alert/custom-values.yaml +++ /dev/null @@ -1,305 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmalert-supply-prd - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmselect-supply-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vminsert-supply-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmselect-supply-prd-proxy.victoriametrics.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `notifiers` section - notifier: - alertmanager: - url: "http://alertmanager-infra-prd.alertmanager.svc.clusterset.local:9093" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config1/supply/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-supply-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-supply-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-cluster-alert/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-cluster-alert/custom-values.yaml deleted file mode 100644 index 9d5b921..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-cluster-alert/custom-values.yaml +++ /dev/null @@ -1,305 +0,0 @@ -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # mount API token to pod directly - automountToken: true - -imagePullSecrets: [] - -dedicatedValue: false -schedulerName: default-scheduler - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - namespaced: false - extraLabels: {} - annotations: {} - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmalert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmalert - -alertmanager: - enabled: false - -server: - enabled: true - name: vmalert - image: - repository: victoriametrics/vmalert - tag: "" # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - nameOverride: "" - fullnameOverride: vmcluster-alert - configMap: "" - - ## See `kubectl explain poddisruptionbudget.spec` for more - ## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - - replicaCount: 1 - - # deployment strategy, set to standard k8s default - strategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - - # specifies the minimum number of seconds for which a newly created Pod should be ready without any of its containers crashing/terminating - # 0 is the standard k8s default - minReadySeconds: 0 - - # vmalert reads metrics from source, next section represents its configuration. It can be any service which supports - # MetricsQL or PromQL. - datasource: - url: "http://vmcluster-supply-prd-vmselect.victoriametrics-cluster.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for datasource - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for datasource - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - remote: - write: - url: "http://vmcluster-supply-prd-vminsert.victoriametrics-cluster.svc.cluster.local:8480/insert/100/prometheus/" - # -- Basic auth for remote write - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote write - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - read: - url: "http://vmcluster-supply-prd-vmselect.victoriametrics-cluster.svc.cluster.local:8481/select/100/prometheus" - # -- Basic auth for remote read - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for remote read - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Notifier to use for alerts. - # Multiple notifiers can be enabled by using `snotifiers` section - notifier: - alertmanager: - url: "http://dummy-url.com" - # -- Basic auth for alertmanager - basicAuth: - username: "" - password: "" - # -- Auth based on Bearer token for alertmanager - bearer: - # -- Token with Bearer token. You can use one of token or tokenFile. You don't need to add "Bearer" prefix string - token: "" - # -- Token Auth file with Bearer token. You can use one of token or tokenFile - tokenFile: "" - - # -- Additional notifiers to use for alerts - notifiers: [] - # - alertmanager: - # url: "http://devops-p-alertmanager-01b.meeshoint.in:9093" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - # - alertmanager: - # url: "https://prd-infra-alertmanager.meesho.com" - # basicAuth: - # username: "" - # password: "" - # bearer: - # token: "" - # tokenFile: "" - - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - rule: "/config/vm-alerts-config/**/**/*.yaml" - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - extraContainers: - [] - #- name: config-reloader - # image: reloader-image - - service: - annotations: {} - labels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8880 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 - - ingress: - enabled: true - annotations: {} - ingressClassName: nginx-internal - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmalert-supply-prd.meeshogcp.in - path: / - port: http - - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - podSecurityContext: {} - # fsGroup: 2000 - - securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - - resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 500m - memory: 1Gi - - # Annotations to be added to the deployment - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - - # labels to be added to the deployment, pods and other resources - labels: - bu: "infra" - team: "sre" - service: "vmalert-supply-prd" - env: "prd" - priority: "p0" - type: "vmalert" - - # Annotations to be added to pod - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8880" - cluster-autoscaler.kubernetes.io/safe-to-evict: 'false' - - podLabels: {} - - nodeSelector: - dedicated: "vmselect-mds" - - - priorityClassName: "" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - affinity: {} - - # vmalert alert rules configuration configuration: - # use existing configmap if specified - # otherwise .config values will be used - config: - alerts: - groups: [] - - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s -# -- Commented. HTTP scheme to use for scraping. -# scheme: https -# -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true diff --git a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-cluster/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-cluster/custom-values.yaml deleted file mode 100644 index 6655d08..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-cluster/custom-values.yaml +++ /dev/null @@ -1,1377 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "vmcluster" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Override default `app` label name - name: "vmselect" - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # override Chart.AppVersion - tag: v1.132.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMSelect http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: "" - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - # -- Extra command line arguments for vmselect component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8481 - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "supply" - team: "sre" - service: "vmcluster-supply-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # -- Pod's additional labels - podLabels: - bu: "supply" - team: "sre" - service: "vmcluster-supply-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Readiness & Liveness probes - probe: - # -- VMSelect readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 5 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMSelect liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 80 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vmselect - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vmselect - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - dedicated: "vmselect-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 16 - memory: 32Gi - - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: true - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: true - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Create VMSelect service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - - # -- Ingress annotations - annotations: {} - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vmcluster-supply-prd.meeshogcp.in - path: - - /select - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # -- Ingress controller class name - ingressClassName: contour-internal-1 - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for select component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/select" .Values.vmselect }}' - - # -- vmselect mode: deployment, daemonSet - mode: deployment - - # -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables - deployment: - spec: - # -- VMSelect strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - - # -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables - statefulSet: - spec: - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume extra labels - extraLabels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmselect component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- IDs of vmstorage nodes to exclude from writing - excludeStorageIDs: [] - # -- Override default `app` label name - name: "vminsert" - # -- VMInsert strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - # override Chart.AppVersion - tag: v1.132.0-cluster - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMInsert http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: "" - # -- Extra command line arguments for vminsert component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8480 - loggerTimezone: "Asia/Kolkata" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "supply" - team: "sre" - service: "vmcluster-supply-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Pod's additional labels - podLabels: - bu: "supply" - team: "sre" - service: "vmcluster-supply-prd" - env: "prd" - priority: "p0" - type: "vminsert" - terminationGracePeriodSeconds: 30 - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - - # -- Readiness & Liveness probes - probe: - # -- VMInsert readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 5 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMInsert liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert startup probe - startup: {} - - # -- Relabel configuration - relabel: - enabled: false - config: [] - # -- Use existing configmap if specified - # otherwise .config values will be used. Relabel config **should** reside under `relabel.yml` key - configMap: "" - - # Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 2 - # -- Metric for HPA to use to scale the vminsert component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vminsert - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vminsert - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - dedicated: "vminsert-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 4 - memory: 8Gi - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - service: - # -- Create VMInsert service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here]( https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) for details - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: true - - # -- Ingress annotations - annotations: {} - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vmcluster-supply-prd.meeshogcp.in - path: - - /insert - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - - # -- Ingress controller class name - ingressClassName: contour-internal-1 - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for insert component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/insert" .Values.vminsert }}' - - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vminsert component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vmauth: - # -- Enable deployment of vmauth component. - # With vmauth enabled please set `service.clusterIP: None` and `service.type: ClusterIP` for `vminsert` and `vmselect` to use vmauth balancing benefits. - enabled: false - # -- Override default `app` label name - name: "" - # -- VMAuth configuration secret name - configSecretName: "" - # -- VMAuth configuration object. - # - # It's possible to use given below predefined variables in config: - # * `{{ .vm.read }}` - parsed vmselect URL - # * `{{ .vm.write }}` - parsed vminsert URL - # - # Example - # config: - # unauthorized_user: - # url_map: - # - src_paths: - # - '{{ .vm.read.path }}/.*' - # url_prefix: - # - '{{ urlJoin (omit .vm.read "path") }}/' - config: {} - # -- VMAuth Deployment strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image registry - registry: "" - # -- Image repository - repository: victoriametrics/vmauth - # -- Image tag - # override Chart.AppVersion - tag: "" - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMAuth http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmauth component - fullnameOverride: "" - # -- Extra command line arguments for vmauth component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8427 - # -- VMAuth annotations - annotations: {} - # -- VMAuth additional labels - extraLabels: {} - # -- VMAuth pod labels - podLabels: {} - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - # Readiness & Liveness probes - probe: - # -- VMAuth readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 5 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMAuth liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMAuth startup probe - startup: {} - - # Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vmauth component - enabled: false - # -- Maximum replicas for HPA to use to to scale the vmauth component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmauth component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmauth component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - - # -- Extra containers to run in a pod with vmauth - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vmauth - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: [] - # - key: "key" - # operator: "Equal|Exists" - # value: "value" - # effect: "NoSchedule|PreferNoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: {} - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: [] - # -- Pod's annotations - podAnnotations: {} - # -- Count of vmauth pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: {} - # limits: - # cpu: 50m - # memory: 64Mi - # requests: - # cpu: 50m - # memory: 64Mi - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - service: - # -- Create VMAuth service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Details are [here]( https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8427 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. - # Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vmauth component - enabled: false - # -- Ingress annotations - annotations: {} - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmauth.local - path: - - /insert - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmauth-ingress-tls - # hosts: - # - vmauth.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for vmauth component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/" .Values.vmauth }}' - - serviceMonitor: - # -- Enable deployment of Service Monitor for vmauth component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmauth component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmauth component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- Override default `app` label name - name: "vmstorage" - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.132.0-cluster - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMStorage http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: - - # -- Additional environment variables (ex.: secret tokens, flags). Check https://docs.victoriametrics.com/victoriametrics/#environment-variables for details - env: [] - # -- Specify alternative source for env variables - envFrom: [] - # -- Data retention period. Possible units character: h(ours), d(ays), w(eeks), y(ears), if no unit character specified - month. The minimum retention period is 24h. - retentionPeriod: 30d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8482 - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage-sale-24aug" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage-sale-24aug" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - # -- Use an alternate scheduler, e.g. "stork". Check https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ for details - # schedulerName: - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Override Persistent Volume Claim name - name: vmstorage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are https://kubernetes.io/docs/concepts/storage/persistent-volumes/ - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume extra labels - extraLabels: {} - # -- Storage class name. Will be empty if not set - storageClassName: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume - size: 7400Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "supply" - team: "sre" - service: "vmcluster-supply-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - # -- Pod's additional labels - podLabels: - bu: "supply" - team: "sre" - service: "vmcluster-supply-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Details are https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 24 - memory: 420Gi - - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: - enabled: false - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - podSecurityContext: - enabled: false - service: - enabled: true - # -- Service annotations - annotations: {} - # -- Service ClusterIP - clusterIP: None - # -- Service type - type: ClusterIP - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmselect - vmselectPort: 8401 - # -- Extra service ports - extraPorts: [] - # -- Health check node port for a service. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilyPolicy: "" - # -- List of service IP families. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilies: [] - # -- Traffic Distribution. Check https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution - trafficDistribution: "" - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - minReadySeconds: 5 - # -- Readiness probes - probe: - # -- VMStorage readiness probe - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMStorage startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmstorage component - enabled: false - # -- Maximum replicas for HPA to use to to scale the vmstorage component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmstorage component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmstorage component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: - scaleDown: - selectPolicy: Disabled - - vmbackupmanager: - # -- Enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enabled: false - image: - # -- VMBackupManager image registry - registry: "" - # -- VMBackupManager image repository - repository: victoriametrics/vmbackupmanager - # -- VMBackupManager image tag - # override Chart.AppVersion - tag: "" - # -- Variant of the image tag to use. - # e.g. enterprise. - variant: cluster - # -- Disable hourly backups - disableHourly: false - # -- Disable daily backups - disableDaily: false - # -- Disable weekly backups - disableWeekly: false - # -- Disable monthly backups - disableMonthly: false - # -- Backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "" - # -- Backups' retention settings - retention: - # -- Keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 2 - # -- Keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 2 - # -- Keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 2 - # -- Keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 2 - # -- Extra command line arguments for container of component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - # -- Allows to enable restore options for pod. Check [here](https://docs.victoriametrics.com/victoriametrics/vmbackupmanager/#restore-commands) for details - restore: - onStart: - enabled: false - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details - env: [] - # -- Readiness & Liveness probes - probe: - # -- VMBackupManager readiness probe - readiness: - httpGet: - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 5 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMBackupManager liveness probe - liveness: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMBackupManager startup probe - startup: {} - # -- Extra secret mounts for vmbackupmanager - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmstorage component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-insert/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index 8911dc1..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,234 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-supply-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-supply-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "supply" - team: "supply-sre" - service: "vminsert-supply-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 20 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 8 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vminsert-mds" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 5 - memory: 8Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-supply-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-select-rs-test1/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-select-rs-test1/custom-values.yaml deleted file mode 100644 index 62c335d..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-select-rs-test1/custom-values.yaml +++ /dev/null @@ -1,293 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-test-supply-prd - replicaCount: 1 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-test-supply-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 384 - clusternative.maxConcurrentRequests: 384 - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "supply" - team: "supply-sre" - service: "vmselect-test-supply-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 1 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 1 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "vmselect-mds" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 1 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 16 - memory: 32Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: contour-internal-0 - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-test-supply.prd.meesho.int - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-select/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index 7d2fefb..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,293 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-supply-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-supply-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - search.maxConcurrentRequests: 384 - clusternative.maxConcurrentRequests: 384 - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "supply" - team: "supply-sre" - service: "vmselect-supply-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 15 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "vmselect-mds" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 16 - memory: 32Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: contour-internal-0 - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-supply.prd.meesho.int - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-storage/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index d8d4a47..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,316 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-supply-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage-sale-24aug" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage-sale-24aug" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 700Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "supply" - team: "supply-sre" - service: "vmstorage-supply-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 40 - memory: 480Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.91.3-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-demand-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-supply-prd-ase1/victoriametrics-agent/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoriametrics-agent/custom-values.yaml deleted file mode 100644 index 69dc02b..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoriametrics-agent/custom-values.yaml +++ /dev/null @@ -1,485 +0,0 @@ -# -- Meesho orginazation specific fields for victoria-metrics-agent chart. -custom: - configFileFolderPath: "configmap" - scrapeConfigFileName: "supply-scrape.yaml" - -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - cluster: - # -- K8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - dnsDomain: cluster.local. - -# -- Replica count -replicaCount: 2 - -# -- Specify pod lifecycle -lifecycle: {} - -# -- Use an alternate scheduler, e.g. "stork". Check details [here](https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/) -schedulerName: "" - -# -- VMAgent mode: daemonSet, deployment, statefulSet -mode: deployment - -# -- [K8s DaemonSet](https://kubernetes.io/docs/concepts/workloads/controllers/daemonset/) specific variables -daemonSet: - spec: {} - -# -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables -deployment: - spec: - minReadySeconds: 1200 - progressDeadlineSeconds: 1800 - # -- Deployment strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy) for details - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables -statefulSet: - # -- create cluster of vmagents. Check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) - # available since [v1.77.2](https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2) - clusterMode: true - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - spec: - # -- StatefulSet update strategy. Check [here](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies) for details. - updateStrategy: {} - # type: RollingUpdate - -image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - # -- Image tag, set to `Chart.AppVersion` by default - tag: v1.133.0 # rewrites Chart.AppVersion - # -- Variant of the image to use. - # e.g. enterprise, scratch - variant: "" - # -- Image pull policy - pullPolicy: IfNotPresent - -# -- Image pull secrets -imagePullSecrets: [] - -# -- Add additional DNS entries to pods hosts file. Check [official documentation](https://kubernetes.io/docs/tasks/network/customize-hosts-file-for-pods/) -hostAliases: [] -# - ip: 192.168.1.1 -# hostNames: -# - test.example.com -# - another.example.net - -# -- Override chart name -nameOverride: "" - -# -- Override resources fullname -fullnameOverride: "vm-agent-supply-prd" - -# -- Container working directory -containerWorkingDir: "/" - -rbac: - # -- Enables Role/RoleBinding creation - create: true - - # -- Role/RoleBinding annotations - annotations: {} - - # -- Role/RoleBinding labels - extraLabels: {} - - # -- If true and `rbac.enabled`, will deploy a Role/RoleBinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - - # -- additional rules for a role - extraRules: [] - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - # -- Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-supl-sre-vmagnt-prd-mds@meesho-supply-prd-0622.iam.gserviceaccount.com - } - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - # -- mount API token to pod directly - automountToken: true - -# -- See `kubectl explain poddisruptionbudget.spec` for more or check [official documentation](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# -- Generates `remoteWrite.*` flags and config maps with value content for values, that are of type list of map. -# Each item should contain `url` param to pass validation. -remoteWrite: - - url: http://vm-insert-supply-prd.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write - disableOnDiskQueue: false - dropSamplesOnOverload: false - - url: http://prd-census-server-supply.prd-census-server-supply.svc.cluster.local/api/v1/write - disableOnDiskQueue: true - dropSamplesOnOverload: true -# urlRelabelConfig: -# - action: keep -# source_labels: [env] -# regex: "dev" - -# -- VMAgent extra command line arguments -extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8429 - promscrape.config.strictParse: false # default is true - promscrape.maxScrapeSize: 1000000000 # default is 167772160 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - # promscrape.suppressScrapeErrors: true # default is false - - # Uncomment and specify the port if you want to support any of the protocols: - # https://docs.victoriametrics.com/victoriametrics/vmagent/#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for more details. -env: - - name: GOGC - value: "200" - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth-secret - # key: password - -# -- Specify alternative source for env variables -envFrom: - [] - #- configMapRef: - # name: special-config - -# -- Extra labels for Deployment and Statefulset -extraLabels: - bu: "supply" - team: "supply-sre" - service: "vm-agent-supply-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra labels for Pods only -podLabels: - bu: "supply" - team: "supply-sre" - service: "vm-agent-supply-prd" - env: "prd" - priority: "p0" - type: "vmagent" - -# -- Extra selector labels common for pod and service -selectorLabels: {} - -# -- Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# -- Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# -- Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -# -- Extra containers to run in a pod with vmagent -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -# -- Init containers for vmagent -initContainers: - [] - # - name: example - # image: example-image - -# -- Security context to be added to pod -podSecurityContext: - enabled: true - # fsGroup: 2000 - -# -- Security context to be added to pod's containers -securityContext: - enabled: true - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - # -- Enable agent service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - extraLabels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) for details - externalIPs: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8429 - # -- Target port - targetPort: http - # nodePort: 30000 - # -- Service type - type: ClusterIP - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service internal traffic policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/service/#internal-traffic-policy) for details - internalTrafficPolicy: "" - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - # -- Extra selector labels common for service only - selectorLabels: {} - -ingress: - # -- Enable deployment of ingress for agent - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-agent-supply-prd.meeshogcp.in - path: - - / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - - # -- Ingress controller class name - ingressClassName: "nginx-internal" - - # -- Ingress path type - pathType: Prefix - -# -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) -resources: - requests: - cpu: 70 - memory: 100Gi - -# -- Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) -nodeSelector: - dedicated: "vmagent-n4" - -# -- Node tolerations for server scheduling to nodes with taints. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent-n4" - effect: "NoSchedule" - -# -- Pod topologySpreadConstraints -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# -- Pod affinity -affinity: {} - -# -- VMAgent [scraping configuration](https://docs.victoriametrics.com/victoriametrics/vmagent/#how-to-collect-metrics-in-prometheus-format) -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vm-agent-supply-prd" - -# -- Priority class to be assigned to the pod(s) -priorityClassName: "" - -# -- Enable the host network -hostNetwork: false - -serviceMonitor: - # -- Enable deployment of Service Monitor for server component. This is Prometheus operator object - enabled: false - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Service Monitor relabelings - relabelings: [] - # -- Basic auth params for Service Monitor - basicAuth: {} - # -- Service Monitor metricRelabelings - metricRelabelings: [] - # -- Service Monitor targetPort - targetPort: http - # interval: 15s - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - -# -- Empty dir configuration for a case, when persistence is disabled -emptyDir: {} - -persistentVolume: - # -- Create/use Persistent Volume Claim for server component. Empty dir if false - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- StorageClass to use for persistent volume. Requires server.persistentVolume.enabled: true. If defined, PVC created automatically - storageClassName: "" - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Size of the volume. Should be calculated based on the logs you send and retention policy you set. - size: 10Gi - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume additional labels - extraLabels: {} - - # -- Existing Claim name. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Horizontal Pod Autoscaling. -# Note that it is not intended to be used for vmagents which perform scraping. -# In order to scale scraping vmagents check [here](https://docs.victoriametrics.com/victoriametrics/vmagent/#scraping-big-number-of-targets) -horizontalPodAutoscaling: - # -- Use HPA for vmagent - enabled: false - # -- Maximum replicas for HPA to use to to scale vmagent - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale vmagent - minReplicas: 1 - # -- Metric for HPA to use to scale vmagent - metrics: [] - -# -- VMAgent scrape configuration -config: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -probe: - # -- Readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - # -- Liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - # -- Startup probe - startup: {} - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -allowedMetricsEndpoints: - - /metrics - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-supply-prd-ase1/victoriametrics-insert-ht/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoriametrics-insert-ht/custom-values.yaml deleted file mode 100644 index d509d15..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoriametrics-insert-ht/custom-values.yaml +++ /dev/null @@ -1,409 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: false - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- IDs of vmstorage nodes to exclude from writing - excludeStorageIDs: [] - # -- Override default `app` label name - name: vminsert - # -- VMInsert strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - # override Chart.AppVersion - tag: v1.133.0-cluster - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMInsert http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-ht-supply-prd - # -- Extra command line arguments for vminsert component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8480 - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-ht-supply-prd-0.vm-storage-ht-supply-prd.victoriametrics.svc:8400" - - "vm-storage-ht-supply-prd-1.vm-storage-ht-supply-prd.victoriametrics.svc:8400" - - "vm-storage-ht-supply-prd-2.vm-storage-ht-supply-prd.victoriametrics.svc:8400" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "supply" - team: "supply-sre" - service: "vm-insert-ht-supply-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Pod's additional labels - podLabels: - bu: "supply" - team: "supply-sre" - service: "vm-insert-ht-supply-prd" - env: "prd" - priority: "p0" - type: "vminsert" - terminationGracePeriodSeconds: 30 - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - - # -- Readiness & Liveness probes - probe: - # -- VMInsert readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert startup probe - startup: {} - - # -- Relabel configuration - relabel: - enabled: false - config: [] - # -- Use existing configmap if specified - # otherwise .config values will be used. Relabel config **should** reside under `relabel.yml` key - configMap: "" - - # Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 2 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vminsert - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vminsert - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vm-stack-ht" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - dedicated: "vm-stack-ht" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vm-stack-ht - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vm-stack-ht - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 3 - memory: 8Gi - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - service: - # -- Create VMInsert service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here]( https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) for details - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-insert-ht-supply-prd.meeshogcp.in - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - - # -- Ingress controller class name - ingressClassName: nginx-internal - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for insert component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/insert" .Values.vminsert }}' - - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vminsert component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-supply-prd-ase1/victoriametrics-insert/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoriametrics-insert/custom-values.yaml deleted file mode 100644 index cda366f..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoriametrics-insert/custom-values.yaml +++ /dev/null @@ -1,411 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: false - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- IDs of vmstorage nodes to exclude from writing - excludeStorageIDs: [] - # -- Override default `app` label name - name: vminsert - # -- VMInsert strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - # override Chart.AppVersion - tag: v1.107.0-cluster - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMInsert http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vm-insert-supply-prd - # -- Extra command line arguments for vminsert component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8480 - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-supply-prd-0.vm-storage-supply-prd.victoriametrics.svc:8400" - - "vm-storage-supply-prd-1.vm-storage-supply-prd.victoriametrics.svc:8400" - - "vm-storage-supply-prd-2.vm-storage-supply-prd.victoriametrics.svc:8400" - - "vm-storage-supply-prd-3.vm-storage-supply-prd.victoriametrics.svc:8400" - - "vm-storage-supply-prd-4.vm-storage-supply-prd.victoriametrics.svc:8400" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "supply" - team: "supply-sre" - service: "vm-insert-supply-prd" - env: "prd" - priority: "p0" - type: "vminsert" - # -- Pod's additional labels - podLabels: - bu: "supply" - team: "supply-sre" - service: "vm-insert-supply-prd" - env: "prd" - priority: "p0" - type: "vminsert" - terminationGracePeriodSeconds: 30 - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - - # -- Readiness & Liveness probes - probe: - # -- VMInsert readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMInsert startup probe - startup: {} - - # -- Relabel configuration - relabel: - enabled: false - config: [] - # -- Use existing configmap if specified - # otherwise .config values will be used. Relabel config **should** reside under `relabel.yml` key - configMap: "" - - # Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 30 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 8 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vminsert - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vminsert - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - dedicated: "vminsert-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 8 - memory: 8Gi - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - service: - # -- Create VMInsert service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here]( https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Enable UDP port. used if you have `spec.opentsdbListenAddr` specified - # Make sure that service is not type `LoadBalancer`, as it requires `MixedProtocolLBService` feature gate. Check [here](https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/) for details - udp: false - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-insert-supply-prd.meeshogcp.in - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - - # -- Ingress controller class name - ingressClassName: nginx-internal - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for insert component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/insert" .Values.vminsert }}' - - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vminsert component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-supply-prd-ase1/victoriametrics-select-ht/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoriametrics-select-ht/custom-values.yaml deleted file mode 100644 index 26731cf..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoriametrics-select-ht/custom-values.yaml +++ /dev/null @@ -1,451 +0,0 @@ - -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Enable split services for vmselect (extra Service objects will be created by templates/service-split.yaml) - splitService: true - externalService: - enabled: false - name: vmselect-ht-supply-prd-proxy - - # -- Enable clusternative service for LoadBalancer vmselect component - clusternativeService: - enabled: true - port: 8401 - sessionAffinity: None - ipFamilyPolicy: SingleStack - - # -- Override default `app` label name - name: "" - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # override Chart.AppVersion - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMSelect http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-ht-supply-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - # -- Extra command line arguments for vmselect component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8481 - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 180s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-ht-supply-prd-0.vm-storage-ht-supply-prd.victoriametrics.svc:8401" - - "vm-storage-ht-supply-prd-1.vm-storage-ht-supply-prd.victoriametrics.svc:8401" - - "vm-storage-ht-supply-prd-2.vm-storage-ht-supply-prd.victoriametrics.svc:8401" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "supply" - team: "supply-sre" - service: "vm-select-ht-supply-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # -- Pod's additional labels - podLabels: - bu: "supply" - team: "supply-sre" - service: "vm-select-ht-supply-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Readiness & Liveness probes - probe: - # -- VMSelect readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 15 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vmselect - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vmselect - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vm-stack-ht" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - dedicated: "vm-stack-ht" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vm-stack-ht - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vm-stack-ht - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 3 - memory: 32Gi - - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Create VMSelect service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - - # -- Ingress annotations - annotations: {} - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-select-ht-supply-prd.meeshogcp.in - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # -- Ingress controller class name - ingressClassName: contour-internal-0 - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for select component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/select" .Values.vmselect }}' - - # -- vmselect mode: deployment, daemonSet - mode: deployment - - # -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables - deployment: - spec: - # -- VMSelect strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - - # -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables - statefulSet: - enabled: false - spec: - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume extra labels - extraLabels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmselect component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: false - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - diff --git a/helm-overrides/k8s-supply-prd-ase1/victoriametrics-select/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoriametrics-select/custom-values.yaml deleted file mode 100644 index e1eff04..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoriametrics-select/custom-values.yaml +++ /dev/null @@ -1,449 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - - splitService: true - externalService: - enabled: true - name: vmselect-supply-prd-proxy - - # -- Enable clusternative service for LoadBalancer vmselect component - clusternativeService: - enabled: false - - # -- Override default `app` label name - name: "" - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # override Chart.AppVersion - tag: v1.133.0-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Variant of the image to use. - # e.g. cluster, enterprise-cluster - variant: cluster - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMSelect http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vm-select-supply-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppressStorageFQDNsRender: false - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - # -- Extra command line arguments for vmselect component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - httpListenAddr: :8481 - clusternativeListenAddr: ":8401" - dedup.minScrapeInterval: 60s - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - storageNode: - - "vm-storage-supply-prd-0.vm-storage-supply-prd.victoriametrics.svc:8401" - - "vm-storage-supply-prd-1.vm-storage-supply-prd.victoriametrics.svc:8401" - - "vm-storage-supply-prd-2.vm-storage-supply-prd.victoriametrics.svc:8401" - - "vm-storage-supply-prd-3.vm-storage-supply-prd.victoriametrics.svc:8401" - - "vm-storage-supply-prd-4.vm-storage-supply-prd.victoriametrics.svc:8401" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "supply" - team: "supply-sre" - service: "vm-select-supply-prd" - env: "prd" - priority: "p0" - type: "vmselect" - # -- Pod's additional labels - podLabels: - bu: "supply" - team: "supply-sre" - service: "vm-select-supply-prd" - env: "prd" - priority: "p0" - type: "vmselect" - - # -- Additional environment variables (ex.: secret tokens, flags). Check [here](https://docs.victoriametrics.com/victoriametrics/#environment-variables) for details. - env: [] - - # -- Specify alternative source for env variables - envFrom: [] - #- configMapRef: - # name: special-config - - # -- Readiness & Liveness probes - probe: - # -- VMSelect readiness probe - readiness: - httpGet: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect liveness probe - liveness: - tcpSocket: {} - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - # -- VMSelect startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 40 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 15 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - # -- Behavior settings for scaling by the HPA - behavior: {} - - # -- Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # -- Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # -- Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - # -- Extra containers to run in a pod with vmselect - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - # -- Init containers for vmselect - initContainers: - [] - # - name: example - # image: example-image - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Details are [here](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Details are [here](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect-mds" - effect: "NoSchedule" - - # -- Pod's node selector. Details are [here](https://kubernetes.io/docs/concepts/scheduling-eviction/assign-pod-node/#nodeselector) - nodeSelector: - dedicated: "vmselect-mds" - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object. Details are [here](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 18 - memory: 32Gi - - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: - enabled: false - # -- Pod's security context. Details are [here](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - podSecurityContext: - enabled: false - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Create VMSelect service - enabled: true - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service external IPs. Details are [here](https://kubernetes.io/docs/concepts/services-networking/service/#external-ips) - externalIPs: [] - # -- Extra service ports - extraPorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - # -- Health check node port for a service. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check [here](https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip) for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilyPolicy: "" - # -- List of service IP families. Check [here](https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services) for details. - ipFamilies: [] - # -- Traffic Distribution. Check [Traffic distribution](https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution) - trafficDistribution: "" - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: false - - # -- Ingress annotations - annotations: {} - - # -- Ingress extra labels - extraLabels: {} - - # -- Array of host objects - hosts: - - name: vm-select-supply.prd.meesho.int - path: / - port: http - - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - - # -- Ingress controller class name - ingressClassName: contour-internal-0 - - # -- Ingress path type - pathType: Prefix - - route: - # -- Enable deployment of HTTPRoute for select component - enabled: false - # -- HTTPRoute annotations - annotations: {} - # -- HTTPRoute extra labels - labels: {} - # -- HTTPGateway objects refs - parentRefs: [] - # -- Array of hostnames - hostnames: [] - # -- Extra rules to prepend to route. This is useful when working with annotation based services. - extraRules: [] - # -- Filters for a default rule in HTTPRoute - filters: [] - # -- Matches for a default rule in HTTPRoute - matches: - - path: - type: PathPrefix - value: '{{ dig "extraArgs" "http.pathPrefix" "/select" .Values.vmselect }}' - - # -- vmselect mode: deployment, daemonSet - mode: deployment - - # -- [K8s Deployment](https://kubernetes.io/docs/concepts/workloads/controllers/deployment/) specific variables - deployment: - spec: - # -- VMSelect strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - - # -- [K8s StatefulSet](https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/) specific variables - statefulSet: - enabled: false - spec: - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Override Persistent Volume Claim name - name: "" - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Details are [here](https://kubernetes.io/docs/concepts/storage/persistent-volumes/) - accessModes: - - ReadWriteOnce - - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume extra labels - extraLabels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # -- Basic auth params for Service Monitor - basicAuth: {} - # Commented. Prometheus scare interval for vmselect component - # interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component - # scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. - # scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint - # tlsConfig: - # insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] - # -- Service Monitor metricRelabelings - metricRelabelings: [] - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: false - -vmauth: - # -- Enable deployment of vmauth component. - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: false - diff --git a/helm-overrides/k8s-supply-prd-ase1/victoriametrics-storage-ht/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoriametrics-storage-ht/custom-values.yaml deleted file mode 100644 index e0e6e2f..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoriametrics-storage-ht/custom-values.yaml +++ /dev/null @@ -1,363 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - enabled: false - -vminsert: - enabled: false - -vmauth: - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- Override default `app` label name - name: vmstorage - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Variant of the image to use. e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMStorage http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-ht-supply-prd - - # -- Additional environment variables (ex.: secret tokens, flags). Check https://docs.victoriametrics.com/victoriametrics/#environment-variables for details - env: [] - # -- Specify alternative source for env variables - envFrom: [] - # -- Data retention period. Possible units character: h(ours), d(ays), w(eeks), y(ears), if no unit character specified - month. The minimum retention period is 24h. - retentionPeriod: 1d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - httpListenAddr: :8482 - loggerTimezone: "Asia/Kolkata" - dedup.minScrapeInterval: 60s - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vm-stack-ht" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vm-stack-ht" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vm-stack-ht - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vm-stack-ht - - # -- Use an alternate scheduler, e.g. "stork". Check https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ for details - # schedulerName: - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Override Persistent Volume Claim name - name: vm-stack-ht-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are https://kubernetes.io/docs/concepts/storage/persistent-volumes/ - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume extra labels - extraLabels: {} - # -- Storage class name. Will be empty if not set - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume - size: 250Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "supply" - team: "supply-sre" - service: "vm-storage-ht-supply-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - # -- Pod's additional labels - podLabels: - bu: "supply" - team: "supply-sre" - service: "vm-storage-ht-supply-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 3 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Details are https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 5 - memory: 50Gi - - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: - enabled: false - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - podSecurityContext: - enabled: false - service: - enabled: true - # -- Service annotations - annotations: {} - # -- Service ClusterIP - clusterIP: None - # -- Service type - type: ClusterIP - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmselect - vmselectPort: 8401 - # -- Extra service ports - extraPorts: [] - # -- Health check node port for a service. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilyPolicy: "" - # -- List of service IP families. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilies: [] - # -- Traffic Distribution. Check https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution - trafficDistribution: "" - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - minReadySeconds: 5 - # -- Readiness probes - probe: - # -- VMStorage readiness probe - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMStorage startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmstorage component - enabled: false - # -- Maximum replicas for HPA to use to to scale the vmstorage component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmstorage component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmstorage component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: - scaleDown: - selectPolicy: Disabled - - vmbackupmanager: - # -- Enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" \ No newline at end of file diff --git a/helm-overrides/k8s-supply-prd-ase1/victoriametrics-storage/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1/victoriametrics-storage/custom-values.yaml deleted file mode 100644 index 94ee972..0000000 --- a/helm-overrides/k8s-supply-prd-ase1/victoriametrics-storage/custom-values.yaml +++ /dev/null @@ -1,363 +0,0 @@ -# Default values for victoria-metrics. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -global: - # -- Image pull secrets, that can be shared across multiple helm charts - imagePullSecrets: [] - image: - # -- Image registry, that can be shared across multiple helm charts - registry: "" - vm: - # -- Image tag for all vm charts - tag: "" - # -- Openshift security context compatibility configuration - compatibility: - openshift: - adaptSecurityContext: "auto" - # -- k8s cluster domain suffix, uses for building storage pods' FQDN. Details are [here](https://kubernetes.io/docs/tasks/administer-cluster/dns-custom-nameservers/) - cluster: - dnsDomain: cluster.local. - -# -- Print information after deployment -printNotes: true - -# -- use SRV discovery for storageNode and selectNode flags for enterprise version -autoDiscovery: false - -common: - # -- common for all components image configuration - image: - tag: "" - -serviceAccount: - # -- Specifies whether a service account should be created - create: true - - # -- The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - - # -- Service account labels - extraLabels: {} - - # -- Service account annotations - annotations: {} - - # -- mount API token to pod directly - automountToken: true - -# -- Override chart name -nameOverride: "" - -extraSecrets: - [] - # - name: secret-remote-storage-keys - # annotations: [] - # labels: [] - # data: | - # credentials: b64_encoded_str - -# -- Add extra specs dynamically to this chart -extraObjects: [] - -vmselect: - enabled: false - -vminsert: - enabled: false - -vmauth: - enabled: false - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- Override default `app` label name - name: vmstorage - image: - # -- Image registry - registry: "" - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.133.0-cluster - # -- Variant of the image to use. e.g. cluster, enterprise-cluster - variant: cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Specify pod lifecycle - lifecycle: {} - ports: - # -- VMStorage http port name - name: "http" - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vm-storage-supply-prd - - # -- Additional environment variables (ex.: secret tokens, flags). Check https://docs.victoriametrics.com/victoriametrics/#environment-variables for details - env: [] - # -- Specify alternative source for env variables - envFrom: [] - # -- Data retention period. Possible units character: h(ours), d(ays), w(eeks), y(ears), if no unit character specified - month. The minimum retention period is 24h. - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: true - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - httpListenAddr: :8482 - loggerTimezone: "Asia/Kolkata" - dedup.minScrapeInterval: 60s - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage-n4d" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage-n4d" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - # -- Use an alternate scheduler, e.g. "stork". Check https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ for details - # schedulerName: - - # -- Empty dir configuration if persistence is disabled - emptyDir: {} - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Override Persistent Volume Claim name - name: vmstack-storage-volume - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Details are https://kubernetes.io/docs/concepts/storage/persistent-volumes/ - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume extra labels - extraLabels: {} - # -- Storage class name. Will be empty if not set - storageClassName: hyperdisk-balanced - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume - size: 7400Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment annotations - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - # -- StatefulSet/Deployment additional labels - extraLabels: - bu: "supply" - team: "supply-sre" - service: "vm-storage-supply-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - # arch: "arm64" - # runpod: "ondemand" - # -- Pod's additional labels - podLabels: - bu: "supply" - team: "supply-sre" - service: "vm-storage-supply-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Details are https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 58 - memory: 475Gi - - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: - enabled: false - # -- Pod's security context. Details are https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - podSecurityContext: - enabled: false - service: - enabled: true - # -- Service annotations - annotations: {} - # -- Service ClusterIP - clusterIP: None - # -- Service type - type: ClusterIP - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmselect - vmselectPort: 8401 - # -- Extra service ports - extraPorts: [] - # -- Health check node port for a service. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - healthCheckNodePort: "" - # -- Service external traffic policy. Check https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip for details - externalTrafficPolicy: "" - # -- Service IP family policy. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilyPolicy: "" - # -- List of service IP families. Check https://kubernetes.io/docs/concepts/services-networking/dual-stack/#services for details - ipFamilies: [] - # -- Traffic Distribution. Check https://kubernetes.io/docs/concepts/services-networking/service/#traffic-distribution - trafficDistribution: "" - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - minReadySeconds: 5 - # -- Readiness probes - probe: - # -- VMStorage readiness probe - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - # -- VMStorage startup probe - startup: {} - - horizontalPodAutoscaler: - # -- Use HPA for vmstorage component - enabled: false - # -- Maximum replicas for HPA to use to to scale the vmstorage component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vmstorage component - minReplicas: 2 - # -- Metric for HPA to use to scale the vmstorage component - metrics: [] - # -- Behavior settings for scaling by the HPA - behavior: - scaleDown: - selectPolicy: Disabled - - vmbackupmanager: - # -- Enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enabled: false - -# -- Enterprise license key configuration for VictoriaMetrics enterprise. -# Required only for VictoriaMetrics enterprise. Check docs [here](https://docs.victoriametrics.com/victoriametrics/enterprise/), -# for more information, visit [site](https://victoriametrics.com/products/enterprise/). -# Request a trial license [here](https://victoriametrics.com/products/enterprise/trial/) -# Supported starting from VictoriaMetrics v1.94.0 -license: - # -- License key - key: "" - - # -- Use existing secret with license key - secret: - # -- Existing secret name - name: "" - # -- Key in secret with license key - key: "" diff --git a/helm-overrides/k8s-supply-prd-ase1c/conntrack-adjuster/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/conntrack-adjuster/custom-values.yaml deleted file mode 100644 index 7b64b15..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/conntrack-adjuster/custom-values.yaml +++ /dev/null @@ -1,26 +0,0 @@ -daemonSet: - namespace: prd-conntrack-adjuster - -conntrack: - # Maximum number of conntrack entries - max: 2097152 - # Hash size for conntrack - hashsize: 524288 - # Sleep interval between adjustments (seconds) - sleepInterval: 30 - -# Enable additional matchExpressions -# addExtraMatchExpressions: true - -# Additional matchExpressions to append -# additionalMatchExpressions: -# - key: node-role -# operator: In -# values: -# - "worker" -# - "ingress" -# - key: environment -# operator: In -# values: -# - "production" -# - "staging" \ No newline at end of file diff --git a/helm-overrides/k8s-supply-prd-ase1c/contour-cert-checker/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/contour-cert-checker/custom-values.yaml deleted file mode 100644 index e043e22..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/contour-cert-checker/custom-values.yaml +++ /dev/null @@ -1,28 +0,0 @@ -cronJob: - image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/contour-cert-checker - tag: v1.1 - schedule: "* 12 * * *" - args: ["--cluster=k8s-supply-prd-ase1c"] - resources: - requests: - memory: "50Mi" - cpu: "50m" - limits: - memory: "100Mi" - cpu: "100m" - nodeSelector: - dedicated: supply-devops - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: supply-devops - backoffLimit: 3 - historyLimit: - successfulJobs: 7 - failedJobs: 7 - -rbac: - namespace: contour-cert-checker-ns - serviceAccountName: contour-cert-checker-sa \ No newline at end of file diff --git a/helm-overrides/k8s-supply-prd-ase1c/contour-external/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/contour-external/custom-values.yaml deleted file mode 100644 index ee3a59c..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/contour-external/custom-values.yaml +++ /dev/null @@ -1,86 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled - network: - num-trusted-hops: 1 -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: supply - team: supply-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 250m - memory: 2048Mi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external - nodeSelector: - dedicated: contour-external - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-external" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: supply - team: supply-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-external - nodeSelector: - dedicated: contour-external - logLevel: error - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 8Gi - service: - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-ext-supply-c-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-supply-prd-ase1c/contour-internal-0/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/contour-internal-0/custom-values.yaml deleted file mode 100644 index d3af2a3..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/contour-internal-0/custom-values.yaml +++ /dev/null @@ -1,97 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: supply - team: supply-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 4Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - enabled: true - podLabels: - bu: supply - team: supply-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "55" - targetMemory: "55" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 12 - memory: 8Gi - service: - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int0-supply-c-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-supply-prd-ase1c/contour-internal-1/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/contour-internal-1/custom-values.yaml deleted file mode 100644 index c9ef058..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/contour-internal-1/custom-values.yaml +++ /dev/null @@ -1,89 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: supply - team: supply-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 8 - memory: 11Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1-new - nodeSelector: - dedicated: contour-internal-1-new - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: supply - team: supply-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - logLevel: error - autoscaling: - enabled: true - minReplicas: 3 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - tcpLB: true - export: - enabled: true - targetPorts: - http: http - https: https - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "envoy-int1-supply-c-prd"}}}' - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-supply-prd-ase1c/contour-internal-intra-0/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/contour-internal-intra-0/custom-values.yaml deleted file mode 100644 index e212b3e..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/contour-internal-intra-0/custom-values.yaml +++ /dev/null @@ -1,97 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - maxUnavailable: 0 - replicaCount: 3 - podLabels: - bu: supply - team: supply-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 4Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-0" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - image: - tag: 1.28.0-debian-11-r8 - updateStrategy: - type: RollingUpdate - rollingUpdate: - maxSurge: 10% - enabled: true - podLabels: - bu: supply - team: supply-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-0 - nodeSelector: - dedicated: contour-internal-0 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "55" - targetMemory: "55" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 12 - memory: 8Gi - service: - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-supply-prd-ase1c/contour-internal-intra-1/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/contour-internal-intra-1/custom-values.yaml deleted file mode 100644 index c250863..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/contour-internal-intra-1/custom-values.yaml +++ /dev/null @@ -1,87 +0,0 @@ -configInline: - enableExternalNameService: true - timeouts: - connection-idle-timeout: 305s - connection-shutdown-grace-period: 300s - max-connection-duration: 1200s - disablePermitInsecure: false - tls: - fallback-certificate: {} - accesslog-format: envoy - accesslog-level: disabled -contour: - enabled: true - replicaCount: 3 - podLabels: - bu: supply - team: supply-devops - env: prd - manageCRDs: true - resources: - requests: - cpu: 2 - memory: 4Gi - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1-new - nodeSelector: - dedicated: contour-internal-1-new - service: - type: ClusterIP - ports: - xds: 8001 - metrics: 8000 - ingressClass: - name: "contour-internal-intra-1" - create: true - debug: false - podAnnotations: - prometheus.io/path: /metrics - prometheus.io/port: '8000' - prometheus.io/scrape: 'true' -envoy: - enabled: true - podLabels: - bu: supply - team: supply-devops - env: prd - kind: deployment - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: contour-internal-1 - nodeSelector: - dedicated: contour-internal-1 - logLevel: error - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 250 - targetCPU: "50" - targetMemory: "50" - podAnnotations: - prometheus.io/path: /stats/prometheus - prometheus.io/port: '8002' - prometheus.io/scrape: 'true' - resources: - requests: - cpu: 6 - memory: 3Gi - service: - tcpLB: false - export: - enabled: false - targetPorts: - http: http - https: https - type: ClusterIP - ports: - http: 80 - https: 443 - grpc: 8080 ## only when tcpLB is true it will be used - useHostPort: false -defaultBackend: - enabled: false diff --git a/helm-overrides/k8s-supply-prd-ase1c/coredns/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/coredns/custom-values.yaml deleted file mode 100644 index e6d22c0..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/coredns/custom-values.yaml +++ /dev/null @@ -1,30 +0,0 @@ -replicaCount: 16 - -labels: - bu: supply - team: supply-devops - env: prd - -clusterIP: 10.212.16.2 - - -resources: - limits: - cpu: 100m - memory: 128Mi - requests: - cpu: 100m - memory: 128Mi - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "supply-devops" - effect: "NoSchedule" - -nodeSelector: - dedicated: supply-devops - kubernetes.io/os: linux - -overwriteRegion: "c" -communicationType: "" \ No newline at end of file diff --git a/helm-overrides/k8s-supply-prd-ase1c/external-secrets/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/external-secrets/custom-values.yaml deleted file mode 100644 index 6ef1f70..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/external-secrets/custom-values.yaml +++ /dev/null @@ -1,50 +0,0 @@ -external-secrets: - replicaCount: 1 - concurrent: 8 - resources: - requests: - cpu: 100m - memory: 256Mi - - # -- If set, install and upgrade CRDs through helm chart. - installCRDs: true - - crds: - # -- If true, create CRDs for Cluster External Secret. - createClusterExternalSecret: true - # -- If true, create CRDs for Cluster Secret Store. - createClusterSecretStore: true - metrics: - service: - # -- Enable if you use another monitoring tool than Prometheus to scrape the metrics - enabled: false - - # -- Metrics service port to scrape - port: 8080 - - # -- Additional service annotations - annotations: {} - - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: supply-devops - nodeSelector: - dedicated: supply-devops - certController: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: supply-devops - nodeSelector: - dedicated: supply-devops - webhook: - tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: supply-devops - nodeSelector: - dedicated: supply-devops diff --git a/helm-overrides/k8s-supply-prd-ase1c/flagger/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/flagger/custom-values.yaml deleted file mode 100644 index 64e06f9..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/flagger/custom-values.yaml +++ /dev/null @@ -1,69 +0,0 @@ -# Default values for flagger. - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/flagger - tag: rollout-service-v2.0.0 - -# accepted values are debug, info, warning, error (defaults to info) -logLevel: info - -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8080" - appmesh.k8s.aws/sidecarInjectorWebhook: disabled - -crd: - # crd.create: `true` if custom resource definitions should be created - create: false - -resources: - limits: - memory: "1024Mi" - cpu: "2" - requests: - memory: "512Mi" - cpu: "500m" - -nodeSelector: - dedicated: supply-devops - -tolerations: - - effect: NoSchedule - key: dedicated - operator: Equal - value: supply-devops - -prometheus: - install: false - image: docker.io/prom/prometheus:v2.39.1 - pullSecret: - retention: 2h - securityContext: - enabled: false - context: - readOnlyRootFilesystem: true - runAsUser: 10001 - -podDisruptionBudget: - enabled: false - minAvailable: 1 - -podLabels: - env: prd - team: supply-devops - bu: infra - region: ase1c - -env: -- name: ROLLOUT_SERVICE_URL - value: "http://prd-supply-rollout-service.prd-supply-rollout-service.svc.cluster.local" - - -rolloutService: - enabled: true - apps: - prd-devops-node-app-consumers: true - prd-devops-node-app: true - prd-devops-golang-app: true - prd-devops-golang-app-test: true - prd-devops-golang-app-crons: true diff --git a/helm-overrides/k8s-supply-prd-ase1c/fluentd/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/fluentd/custom-values.yaml deleted file mode 100644 index acdde9f..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/fluentd/custom-values.yaml +++ /dev/null @@ -1,720 +0,0 @@ -nameOverride: "" -fullnameOverride: "" - -# DaemonSet, Deployment or StatefulSet -kind: "DaemonSet" -# azureblob, cloudwatch, elasticsearch7, elasticsearch8, gcs, graylog , kafka, kafka2, kinesis, opensearch -variant: gcs -# # Only applicable for Deployment or StatefulSet -# replicaCount: 1 - -image: - repository: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/fluentd-v2" - pullPolicy: "Always" - tag: "edge-debian" - -## Optional array of imagePullSecrets containing private registry credentials -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ -imagePullSecrets: [] - -serviceAccount: - create: true - annotations: { - iam.gke.io/gcp-service-account: sa-supl-spsre-fluentd-prd@meesho-supply-ase1c-prd-0625.iam.gserviceaccount.com - } - name: null - -rbac: - create: true - -# from Kubernetes 1.25, PSP is deprecated -# See: https://kubernetes.io/blog/2022/08/23/kubernetes-v1-25-release/#pod-security-changes -# We automatically disable PSP if Kubernetes version is 1.25 or higher -podSecurityPolicy: - enabled: true - annotations: {} - -## Security Context policies for controller pods -## See https://kubernetes.io/docs/tasks/administer-cluster/sysctl-cluster/ for -## notes on enabling and using sysctls -## -podSecurityContext: {} - # seLinuxOptions: - # type: "spc_t" - -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -# Configure the livecycle -# Ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/ -lifecycle: {} - # preStop: - # exec: - # command: ["/bin/sh", "-c", "sleep 20"] - -# Configure the livenessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#livenessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -# Configure the readinessProbe -# Ref: https://kubernetes.io/docs/tasks/configure-pod-container/configure-liveness-readiness-startup-probes/ -#readinessProbe: -# httpGet: -# path: /metrics -# port: metrics - # initialDelaySeconds: 0 - # periodSeconds: 10 - # timeoutSeconds: 1 - # successThreshold: 1 - # failureThreshold: 3 - -resources: - requests: - cpu: 100m - memory: 50Mi - limits: - memory: 2000Mi - cpu: 2000m - -## only available if kind is Deployment -autoscaling: - enabled: false - minReplicas: 1 - maxReplicas: 100 - targetCPUUtilizationPercentage: 80 - # targetMemoryUtilizationPercentage: 80 - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale-walkthrough/#autoscaling-on-multiple-metrics-and-custom-metrics - customRules: [] - # - type: Pods - # pods: - # metric: - # name: packets-per-second - # target: - # type: AverageValue - # averageValue: 1k - ## see https://kubernetes.io/docs/tasks/run-application/horizontal-pod-autoscale/#support-for-configurable-scaling-behavior - # behavior: - # scaleDown: - # policies: - # - type: Pods - # value: 4 - # periodSeconds: 60 - # - type: Percent - # value: 10 - # periodSeconds: 60 - - -priorityClassName: "system-node-critical" - -nodeSelector: {} - -## Node tolerations for server scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - operator: Exists - -## Affinity and anti-affinity -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/#affinity-and-anti-affinity -## -affinity: {} - -## Annotations to be added to fluentd DaemonSet/Deployment -## -annotations: {} - -## Labels to be added to fluentd DaemonSet/Deployment -## -labels: - bu: supply - team: sre - type: fluentd - service: fluentd-supply-prd - priority: p0 - env: prd - zone_extended: ase1c - -## Annotations to be added to fluentd pods -## -podAnnotations: {} - -## Labels to be added to fluentd pods -## -podLabels: - bu: supply - team: sre - type: fluentd - service: fluentd-supply-prd - priority: p0 - env: prd - zone_extended: ase1c - - -## How long (in seconds) a pods needs to be stable before progressing the deployment -## -minReadySeconds: - -## How long (in seconds) a pod may take to exit (useful with lifecycle hooks to ensure lb deregistration is done) -## -terminationGracePeriodSeconds: - -## Deployment strategy / DaemonSet updateStrategy -## -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 25% - maxSurge: 0 - -## Additional environment variables to set for fluentd pods -## Additional environment variables to set for fluentd pods -env: -- name: APP_NAME - value: namespace_name -- name: SUB_SYSTEM - value: container_name -# - name: FLUENTD_CONF -# value: "../../etc/fluent/fluent.conf" -- name: APP_NAME_SYSTEMD - value: systemd -- name: SUB_SYSTEM_SYSTEMD - value: kubelet.service -- name: ENDPOINT - value: ingress.coralogixsg.com -- name: LOG_LEVEL - value: error -- name: TZ - value: "Asia/Kolkata" -- name: K8S_NODE_NAME - valueFrom: - fieldRef: - fieldPath: spec.nodeName - - -externalSecret: - secretStoreRef: - name: vault-backend - path: meesho/prd/cntr/devop/coralogix-keys - -# externalSecret: -# enabled: true -# key: dev/devops/coralogix -# secretStoreRef: -# name: vault-backend - -envFrom: -- secretRef: - name: integrations-privatekey - -initContainers: [] - -## Name of the configMap containing a custom fluentd.conf configuration file to use instead of the default. -# mainConfigMapNameOverride: "" - -## Name of the configMap containing files to be placed under /etc/fluent/config.d/ -## NOTE: This will replace ALL default files in the aforementioned path! -# extraFilesConfigMapNameOverride: "" - -mountVarLogDirectory: true -mountDockerContainersDirectory: true - -volumes: [] -# - name: varlog -# hostPath: -# path: /var/log -# - name: varlibdockercontainers -# hostPath: -# path: /var/lib/docker/containers -# - name: etcfluentd-main -# configMap: -# name: fluentd-main -# defaultMode: 0777 -# - name: etcfluentd-config -# configMap: -# name: fluentd-config -# defaultMode: 0777 - -volumeMounts: [] -# - name: varlog -# mountPath: /var/log -# - name: varlibdockercontainers -# mountPath: /var/lib/docker/containers -# readOnly: true -# - name: etcfluentd-main -# mountPath: /etc/fluent -# - name: etcfluentd-config -# mountPath: /etc/fluent/config.d/ - -## Only available if kind is StatefulSet -## Fluentd persistence -## -persistence: - enabled: false - storageClass: "" - accessMode: ReadWriteOnce - size: 10Gi - -## Fluentd service -## -service: - enabled: true - type: "ClusterIP" - annotations: {} - # loadBalancerIP: - # externalTrafficPolicy: Local - ports: [] - # - name: "forwarder" - # protocol: TCP - # containerPort: 24224 - -## Prometheus Monitoring -## -metrics: - serviceMonitor: - enabled: false - additionalLabels: - release: prometheus-operator - namespace: "" - namespaceSelector: {} - ## metric relabel configs to apply to samples before ingestion. - ## - metricRelabelings: [] - # - sourceLabels: [__name__] - # separator: ; - # regex: ^fluentd_output_status_buffer_(oldest|newest)_.+ - # replacement: $1 - # action: drop - ## relabel configs to apply to samples after ingestion. - ## - relabelings: [] - # - sourceLabels: [__meta_kubernetes_pod_node_name] - # separator: ; - # regex: ^(.*)$ - # targetLabel: nodename - # replacement: $1 - # action: replace - ## Additional serviceMonitor config - ## - # jobLabel: fluentd - # scrapeInterval: 30s - # scrapeTimeout: 5s - # honorLabels: true - - prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] - # - alert: FluentdDown - # expr: up{job="fluentd"} == 0 - # for: 5m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Down" - # description: "{{ $labels.pod }} on {{ $labels.nodename }} is down" - # - alert: FluentdScrapeMissing - # expr: absent(up{job="fluentd"} == 1) - # for: 15m - # labels: - # context: fluentd - # severity: warning - # annotations: - # summary: "Fluentd Scrape Missing" - # description: "Fluentd instance has disappeared from Prometheus target discovery" - -## Grafana Monitoring Dashboard -## -dashboards: - enabled: "true" - namespace: "" - labels: - grafana_dashboard: '"1"' - -## Fluentd list of plugins to install -## -plugins: [] -# - fluent-plugin-out-http - -## Add fluentd config files from K8s configMaps -## -configMapConfigs: [] -# - fluentd-prometheus-conf -# - fluentd-systemd-conf - -## Fluentd configurations: -## -fileConfigs: - 01_sources.conf: |- - - @type systemd - path /var/log/journal - tag sys-log - read_from_head true - - - @id fluentd-containers.log - @type tail - encoding utf-8 - path /var/log/containers/*.log - pos_file /var/log/containers.log.pos - exclude_path ["/var/log/containers/*telegraf*.log","/var/log/containers/*opentelemetry*.log","/var/log/containers/*fluentbit*.log","/var/log/containers/*gke-metrics*.log","/var/log/containers/*event-exporter-gke*.log","/var/log/containers/*metadata-server*.log","/var/log/containers/*metrics-server*.log","/var/log/containers/*filestore-node*.log"] - path_key filename - tag raw.containers.* - read_from_head true - - @type multi_format - - format json - time_key time - time_format %Y-%m-%dT%H:%M:%S.%NZ - keep_time_key true - - - format /^(? - - format /^(?fsp\s+.+)$/ - ignorecase false - multiline false - - - - - @id raw.containers - @type detect_exceptions - remove_tag_prefix raw - message log - stream stream - multiline_flush_interval 5 - max_bytes 500000 - max_lines 1000 - - - - @type kubernetes_metadata - - - - @type record_transformer - enable_ruby true - - container_id ${record.dig("docker", "container_id")} - - - - - @type rewrite_tag_filter - - key $.kubernetes.namespace_name - pattern ^(.+)$ - tag $1.${tag} - - - - 02_filters.conf: |- - - 03_dispatch.conf: |- - - @type "relabel" - @label @NOCONCATDISPATCH - - - @type "relabel" - @label @CONCATDISPATCH - - - - - - - - - - - # - - 04_outputs.conf: |- diff --git a/helm-overrides/k8s-supply-prd-ase1c/ingress-nginx/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/ingress-nginx/custom-values.yaml deleted file mode 100644 index a7b356a..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/ingress-nginx/custom-values.yaml +++ /dev/null @@ -1,32 +0,0 @@ -ingress-nginx: - controller: - config: - proxy-body-size: "50g" - metrics: - enabled: true - podAnnotations: - prometheus.io/port: "10254" - prometheus.io/scrape: "true" - resources: - requests: - cpu: 200m - memory: 512Mi - autoscaling: - enabled: true - minReplicas: 4 - maxReplicas: 30 - targetCPUUtilizationPercentage: 60 - targetMemoryUtilizationPercentage: 60 - ingressClassResource: - name: nginx-internal - service: - type: ClusterIP - annotations: - cloud.google.com/neg: '{"exposed_ports": {"80":{"name": "nginx-supply-ase1c-prd"}}}' - nodeSelector: - dedicated: devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent" - effect: "NoSchedule" diff --git a/helm-overrides/k8s-supply-prd-ase1c/keda/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/keda/custom-values.yaml deleted file mode 100644 index b5f9a34..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/keda/custom-values.yaml +++ /dev/null @@ -1,35 +0,0 @@ -keda: - operator: - replicaCount: 2 - metricsServer: - replicaCount: 2 - nodeSelector: - dedicated: supply-devops - tolerations: - - key: "dedicated" - operator: "Equal" - value: "supply-devops" - effect: "NoSchedule" - podLabels: - bu: "supply" - team: "supply-devops" - metricsAdapter: - bu: "supply" - team: "supply-devops" - resources: - # -- Manage [resource request & limits] of KEDA operator pod - operator: - limits: - cpu: 1 - memory: 1000Mi - requests: - cpu: 100m - memory: 200Mi - # -- Manage [resource request & limits] of KEDA admission webhooks pod - webhooks: - limits: - cpu: 50m - memory: 150Mi - requests: - cpu: 10m - memory: 80Mi \ No newline at end of file diff --git a/helm-overrides/k8s-supply-prd-ase1c/kube-dns/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/kube-dns/custom-values.yaml deleted file mode 100644 index 59fef7f..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/kube-dns/custom-values.yaml +++ /dev/null @@ -1,2 +0,0 @@ -stubDomains: >- - {"clusterset.local":["169.254.169.254"],"prd.meesho.int.svc.cluster.local":["10.212.16.2"],"prd.mrouter.int.svc.cluster.local":["10.212.16.2"],"mq-server.meeshoint.in.svc.cluster.local":["10.212.16.2"]} diff --git a/helm-overrides/k8s-supply-prd-ase1c/kube-events/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/kube-events/custom-values.yaml deleted file mode 100644 index 2eff5a2..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/kube-events/custom-values.yaml +++ /dev/null @@ -1,149 +0,0 @@ - -fullnameOverride: kube-events-supply-ase1c-prd - -operator: - enabled: true - image: - repository: kubesphere/kube-events-operator - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - configReloader: - image: jimmidyson/configmap-reload:v0.7.1 - affinity: {} - nodeSelector: - dedicated: supply-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: supply-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 200Mi - requests: - cpu: 20m - memory: 20Mi - # Additional volumes on the Deployment definition. - volumes: [] - # Additional volumeMounts on the Deployment definition. - volumeMounts: [] - serviceAccount: - create: true - name: "" - # If true, just clean up cr but not crd - cleanupAllCustomResources: false - kubectlImage: docker.io/bitnami/kubectl:1.14.1 - -exporter: - enabled: true - image: - repository: kubesphere/kube-events-exporter - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: supply-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: supply-devops - operator: Equal - resources: - limits: - cpu: 200m - memory: 500Mi - requests: - cpu: 20m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - sinks: - stdout: - enabled: true - additionalWebhooks: [] - # - url: - # service: - # namespace: - # name: - # port: - # path: - -# Configure fluentbit(operated by https://github.com/fluent/fluent-operator) to collect events logs of exporter. -# These will be applied only when exporter.stdout.enabled=true and fluentbit.enabled=true. -fluentbit: - enabled: false - # Set this to containerd or crio if you want fluentbit to collect CRI format logs. - # If not set, it will be auto detected. - containerRuntime: "" - input: - enabled: true - tail: - refreshIntervalSeconds: 10 - memBufLimit: 5MB - skipLongLines: true - dbSync: Normal - filter: - enabled: true - additionalFilters: [] - output: - enabled: true - opensearch: - host: opensearch-cluster-data.kubesphere-logging-system.svc - port: 9200 - logstashPrefix: ks-whizard-events - suppressTypeName: true - logstashFormat: true - generateID: true - -ruler: - enabled: false - replicas: 2 - image: - repository: kubesphere/kube-events-ruler - tag: "" # If unset use v+ .Chart.appVersion - pullPolicy: IfNotPresent - affinity: {} - nodeSelector: - dedicated: supply-devops - tolerations: - - effect: NoSchedule - key: dedicated - value: supply-devops - operator: Equal - resources: - limits: - cpu: 500m - memory: 500Mi - requests: - cpu: 50m - memory: 50Mi - # Additional volumes on the output Deployment definition. - volumes: [] - # Additional volumeMounts on the output Deployment definition. - volumeMounts: [] - ruleNamespaceSelector: {} - ruleSelector: {} - sinks: - alertmanagers: - - namespace: kubesphere-monitoring-system - name: alertmanager-operated - # webhooks: - # - type: - # url: - # service: - # namespace: - # name: - # port: - # path: - ## 'stdout' sink type can be either 'notification' or 'alert' - # stdout: - # type: notification -rule: - createDefaults: true - overrideDefaults: false - -# Set timezone env variable to be set in containers -timezone: "Asia/Kolkata" diff --git a/helm-overrides/k8s-supply-prd-ase1c/kube-state-metrics/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/kube-state-metrics/custom-values.yaml deleted file mode 100644 index 9117dcb..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/kube-state-metrics/custom-values.yaml +++ /dev/null @@ -1,478 +0,0 @@ -# Default values for kube-state-metrics. -prometheusScrape: true -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/kube-state-metrics - # If unset use v + .Charts.appVersion - tag: v2.9.2 - sha: "" - pullPolicy: IfNotPresent - -fullnameOverride: kube-state-metrics-supply-ase1c-prd - -dedicatedValue: false - -imagePullSecrets: [] -# - name: "image-pull-secret" - -ingress: - enabled: false - ingressClassName: internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: clustermetrics-supply-ase1c-prd.meesho.com - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# If set to true, this will deploy kube-state-metrics as a StatefulSet and the data -# will be automatically sharded across <.Values.replicas> pods using the built-in -# autodiscovery feature: https://github.com/kubernetes/kube-state-metrics#automated-sharding -# This is an experimental feature and there are no stability guarantees. -autosharding: - enabled: false - -replicas: 2 - -# List of additional cli arguments to configure kube-state-metrics -# for example: --enable-gzip-encoding, --log-file, etc. -# all the possible args can be found here: https://github.com/kubernetes/kube-state-metrics/blob/master/docs/cli-arguments.md -extraArgs: [] - -service: - port: 8080 - # Default to clusterIP for backward compatibility - type: ClusterIP - nodePort: 0 - loadBalancerIP: "" - # Only allow access to the loadBalancerIP from these IPs - loadBalancerSourceRanges: [] - clusterIP: "" - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "supply" - team: "supply-sre" - service: "kube-state-metrics-supply-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - - # app: kube-state-metrics - -## Override selector labels -selectorOverride: {} - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -hostNetwork: false - -rbac: - # If true, create & use RBAC resources - create: true - - # Set to a rolename to use existing role - skipping role creating - but still doing serviceaccount and rolebinding to it, rolename set here. - # useExistingRole: your-existing-role - - # If set to false - Run without Cluteradmin privs needed - ONLY works if namespace is also set (if useExistingRole is set this name is used as ClusterRole or Role to bind to) - useClusterRole: true - - # Add permissions for CustomResources' apiGroups in Role/ClusterRole. Should be used in conjunction with Custom Resource State Metrics configuration - # Example: - # - apiGroups: ["monitoring.coreos.com"] - # resources: ["prometheuses"] - # verbs: ["list", "watch"] - extraRules: [] - -# Configure kube-rbac-proxy. When enabled, creates one kube-rbac-proxy container per exposed HTTP endpoint (metrics and telemetry if enabled). -# The requests are served through the same service but requests are then HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - - ## volumeMounts enables mounting custom volumes in rbac-proxy containers - ## Useful for TLS certificates and keys - volumeMounts: [] - # - mountPath: /etc/tls - # name: kube-rbac-proxy-tls - # readOnly: true - -serviceAccount: - # Specifies whether a ServiceAccount should be created, require rbac true - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - # Reference to one or more secrets to be used when pulling images - # ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - imagePullSecrets: [] - # ServiceAccount annotations. - # Use case: AWS EKS IAM roles for service accounts - # ref: https://docs.aws.amazon.com/eks/latest/userguide/specify-service-account-role.html - annotations: {} - -prometheus: - monitor: - enabled: false - annotations: {} - additionalLabels: {} - namespace: "" - jobLabel: "" - targetLabels: [] - podTargetLabels: [] - interval: "" - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - scrapeTimeout: "" - proxyUrl: "" - selectorOverride: {} - honorLabels: false - metricRelabelings: [] - relabelings: [] - scheme: "" - ## File to read bearer token for scraping targets - bearerTokenFile: "" - ## Secret to mount to read bearer token for scraping targets. The secret needs - ## to be in the same namespace as the service monitor and accessible by the - ## Prometheus Operator - bearerTokenSecret: {} - # name: secret-name - # key: key-name - tlsConfig: {} - -## Specify if a Pod Security Policy for kube-state-metrics must be created -## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/ -## -podSecurityPolicy: - enabled: false - annotations: {} - ## Specify pod annotations - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#apparmor - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#seccomp - ## Ref: https://kubernetes.io/docs/concepts/policy/pod-security-policy/#sysctl - ## - # seccomp.security.alpha.kubernetes.io/allowedProfileNames: '*' - # seccomp.security.alpha.kubernetes.io/defaultProfileName: 'docker/default' - # apparmor.security.beta.kubernetes.io/defaultProfileName: 'runtime/default' - - additionalVolumes: [] - -## Configure network policy for kube-state-metrics -networkPolicy: - enabled: false - # networkPolicy.flavor -- Flavor of the network policy to use. - # Can be: - # * kubernetes for networking.k8s.io/v1/NetworkPolicy - # * cilium for cilium.io/v2/CiliumNetworkPolicy - flavor: kubernetes - - ## Configure the cilium network policy kube-apiserver selector - # cilium: - # kubeApiServerSelector: - # - toEntities: - # - kube-apiserver - - # egress: - # - {} - # ingress: - # - {} - # podSelector: - # matchLabels: - # app.kubernetes.io/name: kube-state-metrics - -securityContext: - enabled: true - runAsGroup: 65534 - runAsUser: 65534 - fsGroup: 65534 - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - -## Specify security settings for a Container -## Allows overrides and additional options compared to (Pod) securityContext -## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container -containerSecurityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - -## Node labels for pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -nodeSelector: - dedicated: "supply-devops" - -## Affinity settings for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -affinity: {} - -## Tolerations for pod assignment -## Ref: https://kubernetes.io/docs/concepts/configuration/taint-and-toleration/ -tolerations: - - key: "dedicated" - operator: "Equal" - value: "supply-devops" - effect: "NoSchedule" - -## Topology spread constraints for pod assignment -## Ref: https://kubernetes.io/docs/concepts/workloads/pods/pod-topology-spread-constraints/ -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter - -# Annotations to be added to the deployment/statefulset -annotations: - kubernetes.io/psp: eks.privileged - -# Annotations to be added to the pod -podAnnotations: {} - -## Assign a PriorityClassName to pods if set -# priorityClassName: "" - -# Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: {} - -# Comma-separated list of metrics to be exposed. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricAllowlist: [] - -# Comma-separated list of metrics not to be enabled. -# This list comprises of exact metric names and/or regex patterns. -# The allowlist and denylist are mutually exclusive. -metricDenylist: [] - -# Comma-separated list of additional Kubernetes label keys that will be used in the resource's -# labels metric. By default the metric contains only name and namespace labels. -# To include additional labels, provide a list of resource names in their plural form and Kubernetes -# label keys you would like to allow for them (Example: '=namespaces=[k8s-label-1,k8s-label-n,...],pods=[app],...)'. -# A single '*' can be provided per resource instead to allow any labels, but that has -# severe performance implications (Example: '=pods=[*]'). -metricLabelsAllowlist: - - pods=[*] - - nodes=[*] - - deployments=[*] - - statefulsets=[*] - - persistentvolumeclaims=[*] - - persistentvolumes=[*] - - ingresses=[*] - - namespaces=[*] - - horizontalpodautoscalers=[*] - # - namespaces=[k8s-label-1,k8s-label-n] - -# Comma-separated list of Kubernetes annotations keys that will be used in the resource' -# labels metric. By default the metric contains only name and namespace labels. -# To include additional annotations provide a list of resource names in their plural form and Kubernetes -# annotation keys you would like to allow for them (Example: '=namespaces=[kubernetes.io/team,...],pods=[kubernetes.io/team],...)'. -# A single '*' can be provided per resource instead to allow any annotations, but that has -# severe performance implications (Example: '=pods=[*]'). -metricAnnotationsAllowList: [] - # - pods=[k8s-annotation-1,k8s-annotation-n] - -# Available collectors for kube-state-metrics. -# By default, all available resources are enabled, comment out to disable. -collectors: - - certificatesigningrequests - - configmaps - - cronjobs - - daemonsets - - deployments - - endpoints - - horizontalpodautoscalers - - ingresses - - jobs - - leases - - limitranges - - mutatingwebhookconfigurations - - namespaces - - networkpolicies - - nodes - - persistentvolumeclaims - - persistentvolumes - - poddisruptionbudgets - - pods - - replicasets - - replicationcontrollers - - resourcequotas - - secrets - - services - - statefulsets - - storageclasses - - validatingwebhookconfigurations - - volumeattachments - -# Enabling kubeconfig will pass the --kubeconfig argument to the container -kubeconfig: - enabled: false - # base64 encoded kube-config file - secret: - -# Enabling support for customResourceState, will create a configMap including your config that will be read from kube-state-metrics -customResourceState: - enabled: false - # Add (Cluster)Role permissions to list/watch the customResources defined in the config to rbac.extraRules - config: {} - -# Enable only the release namespace for collecting resources. By default all namespaces are collected. -# If releaseNamespace and namespaces are both set a merged list will be collected. -releaseNamespace: false - -# Comma-separated list(string) or yaml list of namespaces to be enabled for collecting resources. By default all namespaces are collected. -namespaces: "" - -# Comma-separated list of namespaces not to be enabled. If namespaces and namespaces-denylist are both set, -# only namespaces that are excluded in namespaces-denylist will be used. -namespacesDenylist: "" - -## Override the deployment namespace -## -namespaceOverride: "" - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - requests: - cpu: 23m - memory: 169Mi - -## Provide a k8s version to define apiGroups for podSecurityPolicy Cluster Role. -## For example: kubeTargetVersionOverride: 1.14.9 -## -kubeTargetVersionOverride: "" - -# Enable self metrics configuration for service and Service Monitor -# Default values for telemetry configuration can be overridden -# If you set telemetryNodePort, you must also set service.type to NodePort -selfMonitor: - enabled: true - # telemetryHost: 0.0.0.0 - telemetryPort: 8081 - # telemetryNodePort: 0 - -# Enable vertical pod autoscaler support for kube-state-metrics -verticalPodAutoscaler: - enabled: false - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# volumeMounts are used to add custom volume mounts to deployment. -# See example below -volumeMounts: [] -# - mountPath: /etc/config -# name: config-volume - -# volumes are used to add custom volumes to deployment -# See example below -volumes: [] -# - configMap: -# name: cm-for-volume -# name: config-volume diff --git a/helm-overrides/k8s-supply-prd-ase1c/opentelemetry-daemonset/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/opentelemetry-daemonset/custom-values.yaml deleted file mode 100644 index ebf3698..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/opentelemetry-daemonset/custom-values.yaml +++ /dev/null @@ -1,119 +0,0 @@ -fullnameOverride: opentelemetry-supply-ase1c-prd - -mode: daemonset - -priorityClassName: "system-node-critical" - -config: - exporters: - loadbalancing: - routing_key: "traceID" - protocol: - otlp: - timeout: 5s - tls: - insecure: true - sending_queue: - num_consumers: 500 - queue_size: 50000 - resolver: - dns: - hostname: opentelemetry-central-prd-ase1c-headless.opentelemetry.svc.clusterset.local - processors: {} - receivers: - otlp: - protocols: - grpc: - endpoint: ${env:MY_POD_IP}:4317 - http: - endpoint: ${env:MY_POD_IP}:4318 - service: - telemetry: - metrics: - level: detailed - readers: - - pull: - exporter: - prometheus: - host: '0.0.0.0' - port: 8888 - extensions: - - health_check - pipelines: - logs: null - metrics: null - traces: - exporters: - - loadbalancing - processors: [] - receivers: - - otlp - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/opentelemetry-collector-contrib - tag: "0.111.0" - -tolerations: - - operator: Exists - -affinity: - nodeAffinity: - requiredDuringSchedulingIgnoredDuringExecution: - nodeSelectorTerms: - - matchExpressions: - - key: dedicated - operator: NotIn - values: - - vmstorage - - vmselect - - vmagent - - vminsert - - alloy - - contour-internal-0 - - contour-internal-1 - - contour-external - - key: node_pool - operator: NotIn - values: - - np-supply-default-prd-ase1 - -extraEnvs: - - name: K8S_NODE_NAME - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: spec.nodeName - - name: OTEL_RESOURCE_ATTRIBUTES - value: "k8s.node.name=$(K8S_NODE_NAME)" - -ports: - metrics: - enabled: true - -resources: - requests: - cpu: 100m - memory: 124Mi - limits: - cpu: 200m - memory: 256Mi - -podAnnotations: - otel.io/path: "/metrics" - otel.io/scrape: "true" - otel.io/port: "8888" - -podLabels: - bu: "supply" - team: "sre" - service: "opentelemetry-supply-ase1c-prd" - env: "prd" - priority: "p0" - type: "opentelemetry" - arch: "any" - zone_extended: "ase1c" - runpod: "ondemand" - -rollout: - rollingUpdate: - maxUnavailable: 10% \ No newline at end of file diff --git a/helm-overrides/k8s-supply-prd-ase1c/prometheus-node-exporter/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/prometheus-node-exporter/custom-values.yaml deleted file mode 100644 index 586005c..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/prometheus-node-exporter/custom-values.yaml +++ /dev/null @@ -1,498 +0,0 @@ -# Default values for prometheus-node-exporter. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. -image: - registry: quay.io - repository: prometheus/node-exporter - # Overrides the image tag whose default is {{ printf "v%s" .Chart.AppVersion }} - tag: "" - pullPolicy: IfNotPresent - digest: "" - -imagePullSecrets: [] -# - name: "image-pull-secret" -nameOverride: "" -fullnameOverride: "" - -# Number of old history to retain to allow rollback -# Default Kubernetes value is set to 10 -revisionHistoryLimit: 10 - -global: - # To help compatibility with other charts which use global.imagePullSecrets. - # Allow either an array of {name: pullSecret} maps (k8s-style), or an array of strings (more common helm-style). - # global: - # imagePullSecrets: - # - name: pullSecret1 - # - name: pullSecret2 - # or - # global: - # imagePullSecrets: - # - pullSecret1 - # - pullSecret2 - imagePullSecrets: [] - # - # Allow parent charts to override registry hostname - imageRegistry: "" - -# Configure kube-rbac-proxy. When enabled, creates a kube-rbac-proxy to protect the node-exporter http endpoint. -# The requests are served through the same service but requests are HTTPS. -kubeRBACProxy: - enabled: false - image: - registry: quay.io - repository: brancz/kube-rbac-proxy - tag: v0.14.0 - sha: "" - pullPolicy: IfNotPresent - - # List of additional cli arguments to configure kube-rbac-prxy - # for example: --tls-cipher-suites, --log-file, etc. - # all the possible args can be found here: https://github.com/brancz/kube-rbac-proxy#usage - extraArgs: [] - - ## Specify security settings for a Container - ## Allows overrides and additional options compared to (Pod) securityContext - ## Ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-container - containerSecurityContext: {} - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 64Mi - # requests: - # cpu: 10m - # memory: 32Mi - -service: - enabled: true - type: ClusterIP - port: 9100 - targetPort: 9100 - nodePort: - portName: metrics - listenOnAllInterfaces: true - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - ipDualStack: - enabled: false - ipFamilies: ["IPv6", "IPv4"] - ipFamilyPolicy: "PreferDualStack" - -# Set a NetworkPolicy with: -# ingress only on service.port -# no egress permitted -networkPolicy: - enabled: false - -# Additional environment variables that will be passed to the daemonset -env: {} -## env: -## VARIABLE: value - -prometheus: - monitor: - enabled: false - additionalLabels: {} - namespace: "" - - jobLabel: "" - - # List of pod labels to add to node exporter metrics - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#servicemonitor - podTargetLabels: [] - - scheme: http - basicAuth: {} - bearerTokenFile: - tlsConfig: {} - - ## proxyUrl: URL of a proxy that should be used for scraping. - ## - proxyUrl: "" - - ## Override serviceMonitor selector - ## - selectorOverride: {} - - ## Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - ## - attachMetadata: - node: false - - relabelings: [] - metricRelabelings: [] - interval: "" - scrapeTimeout: 10s - ## prometheus.monitor.apiVersion ApiVersion for the serviceMonitor Resource(defaults to "monitoring.coreos.com/v1") - apiVersion: "" - - ## SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - ## - sampleLimit: 0 - - ## TargetLimit defines a limit on the number of scraped targets that will be accepted. - ## - targetLimit: 0 - - ## Per-scrape limit on number of labels that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelLimit: 0 - - ## Per-scrape limit on length of labels name that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelNameLengthLimit: 0 - - ## Per-scrape limit on length of labels value that will be accepted for a sample. Only valid in Prometheus versions 2.27.0 and newer. - ## - labelValueLengthLimit: 0 - - # PodMonitor defines monitoring for a set of pods. - # ref. https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.PodMonitor - # Using a PodMonitor may be preferred in some environments where there is very large number - # of Node Exporter endpoints (1000+) behind a single service. - # The PodMonitor is disabled by default. When switching from ServiceMonitor to PodMonitor, - # the time series resulting from the configuration through PodMonitor may have different labels. - # For instance, there will not be the service label any longer which might - # affect PromQL queries selecting that label. - podMonitor: - enabled: false - # Namespace in which to deploy the pod monitor. Defaults to the release namespace. - namespace: "" - # Additional labels, e.g. setting a label for pod monitor selector as set in prometheus - additionalLabels: {} - # release: kube-prometheus-stack - # PodTargetLabels transfers labels of the Kubernetes Pod onto the target. - podTargetLabels: [] - # apiVersion defaults to monitoring.coreos.com/v1. - apiVersion: "" - # Override pod selector to select pod objects. - selectorOverride: {} - # Attach node metadata to discovered targets. Requires Prometheus v2.35.0 and above. - attachMetadata: - node: false - # The label to use to retrieve the job name from. Defaults to label app.kubernetes.io/name. - jobLabel: "" - - # Scheme/protocol to use for scraping. - scheme: "http" - # Path to scrape metrics at. - path: "/metrics" - - # BasicAuth allow an endpoint to authenticate over basic authentication. - # More info: https://prometheus.io/docs/operating/configuration/#endpoint - basicAuth: {} - # Secret to mount to read bearer token for scraping targets. - # The secret needs to be in the same namespace as the pod monitor and accessible by the Prometheus Operator. - # https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.24/#secretkeyselector-v1-core - bearerTokenSecret: {} - # TLS configuration to use when scraping the endpoint. - tlsConfig: {} - # Authorization section for this endpoint. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.SafeAuthorization - authorization: {} - # OAuth2 for the URL. Only valid in Prometheus versions 2.27.0 and newer. - # https://github.com/prometheus-operator/prometheus-operator/blob/main/Documentation/api.md#monitoring.coreos.com/v1.OAuth2 - oauth2: {} - - # ProxyURL eg http://proxyserver:2195. Directs scrapes through proxy to this endpoint. - proxyUrl: "" - # Interval at which endpoints should be scraped. If not specified Prometheus’ global scrape interval is used. - interval: "" - # Timeout after which the scrape is ended. If not specified, the Prometheus global scrape interval is used. - scrapeTimeout: "" - # HonorTimestamps controls whether Prometheus respects the timestamps present in scraped data. - honorTimestamps: true - # HonorLabels chooses the metric’s labels on collisions with target labels. - honorLabels: true - # Whether to enable HTTP2. Default false. - enableHttp2: "" - # Drop pods that are not running. (Failed, Succeeded). - # Enabled by default. More info: https://kubernetes.io/docs/concepts/workloads/pods/pod-lifecycle/#pod-phase - filterRunning: "" - # FollowRedirects configures whether scrape requests follow HTTP 3xx redirects. Default false. - followRedirects: "" - # Optional HTTP URL parameters - params: {} - - # RelabelConfigs to apply to samples before scraping. Prometheus Operator automatically adds - # relabelings for a few standard Kubernetes fields. The original scrape job’s name - # is available via the __tmp_prometheus_job_name label. - # More info: https://prometheus.io/docs/prometheus/latest/configuration/configuration/#relabel_config - relabelings: [] - # MetricRelabelConfigs to apply to samples before ingestion. - metricRelabelings: [] - - # SampleLimit defines per-scrape limit on number of scraped samples that will be accepted. - sampleLimit: 0 - # TargetLimit defines a limit on the number of scraped targets that will be accepted. - targetLimit: 0 - # Per-scrape limit on number of labels that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelLimit: 0 - # Per-scrape limit on length of labels name that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelNameLengthLimit: 0 - # Per-scrape limit on length of labels value that will be accepted for a sample. - # Only valid in Prometheus versions 2.27.0 and newer. - labelValueLengthLimit: 0 - -## Customize the updateStrategy if set -updateStrategy: - type: RollingUpdate - rollingUpdate: - maxUnavailable: 1 - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - limits: - cpu: 200m - memory: 50Mi - requests: - cpu: 100m - memory: 30Mi - -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is true, a name is generated using the fullname template - name: - annotations: {} - # annotations: { - # iam.gke.io/gcp-service-account: node-exporter-admin-prd@meesho-admin-prd-0622.iam.gserviceaccount.com - # } - imagePullSecrets: [] - automountServiceAccountToken: false - -securityContext: - fsGroup: 65534 - runAsGroup: 65534 - runAsNonRoot: true - runAsUser: 65534 - -containerSecurityContext: - readOnlyRootFilesystem: true - # capabilities: - # add: - # - SYS_TIME - -rbac: - ## If true, create & use RBAC resources - ## - create: true - ## If true, create & use Pod Security Policy resources - ## https://kubernetes.io/docs/concepts/policy/pod-security-policy/ - pspEnabled: true - pspAnnotations: {} - -# for deployments that have node_exporter deployed outside of the cluster, list -# their addresses here -endpoints: [] - -# Expose the service to the host network -hostNetwork: true - -# Share the host process ID namespace -hostPID: true - -# Mount the node's root file system (/) at /host/root in the container -hostRootFsMount: - enabled: true - # Defines how new mounts in existing mounts on the node or in the container - # are propagated to the container or node, respectively. Possible values are - # None, HostToContainer, and Bidirectional. If this field is omitted, then - # None is used. More information on: - # https://kubernetes.io/docs/concepts/storage/volumes/#mount-propagation - mountPropagation: HostToContainer - -## Assign a group of affinity scheduling rules -## -affinity: {} -# nodeAffinity: -# requiredDuringSchedulingIgnoredDuringExecution: -# nodeSelectorTerms: -# - matchFields: -# - key: metadata.name -# operator: In -# values: -# - target-host-name - -# Annotations to be added to node exporter pods -podAnnotations: - # Fix for very slow GKE cluster upgrades - cluster-autoscaler.kubernetes.io/safe-to-evict: "true" - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - - -# Extra labels to be added to node exporter pods -podLabels: - bu: "supply" - team: "supply-sre" - service: "node-exporter-supply-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - zone_extended: "ase1c" - -# Annotations to be added to node exporter daemonset -daemonsetAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9100" - prometheus.io/path: "/metrics" - -## set to true to add the release label so scraping of the servicemonitor with kube-prometheus-stack works out of the box -releaseLabel: false - -# Custom DNS configuration to be added to prometheus-node-exporter pods -dnsConfig: {} -# nameservers: -# - 1.2.3.4 -# searches: -# - ns1.svc.cluster-domain.example -# - my.dns.search.suffix -# options: -# - name: ndots -# value: "2" -# - name: edns0 - -## Assign a nodeSelector if operating a hybrid cluster -## -nodeSelector: {} - # kubernetes.io/os: linux - # kubernetes.io/arch: amd64 - -tolerations: - - operator: Exists - -## Assign a PriorityClassName to pods if set -priorityClassName: "system-node-critical" - -## Additional container arguments -## -extraArgs: [] -# - --collector.diskstats.ignored-devices=^(ram|loop|fd|(h|s|v)d[a-z]|nvme\\d+n\\d+p)\\d+$ -# - --collector.textfile.directory=/run/prometheus - -## Additional mounts from the host to node-exporter container -## -extraHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional configmaps to be mounted. -## -configmaps: [] -# - name: -# mountPath: -secrets: [] -# - name: -# mountPath: -## Override the deployment namespace -## -namespaceOverride: "monitoring" - -## Additional containers for export metrics to text file -## -sidecars: [] -## - name: nvidia-dcgm-exporter -## image: nvidia/dcgm-exporter:1.4.3 - -## Volume for sidecar containers -## -sidecarVolumeMount: [] -## - name: collector-textfiles -## mountPath: /run/prometheus -## readOnly: false - -## Additional mounts from the host to sidecar containers -## -sidecarHostVolumeMounts: [] -# - name: -# hostPath: -# mountPath: -# readOnly: true|false -# mountPropagation: None|HostToContainer|Bidirectional - -## Additional InitContainers to initialize the pod -## -extraInitContainers: [] - -## Liveness probe -## -livenessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -## Readiness probe -## -readinessProbe: - failureThreshold: 3 - httpGet: - httpHeaders: [] - scheme: http - initialDelaySeconds: 0 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - -# Enable vertical pod autoscaler support for prometheus-node-exporter -verticalPodAutoscaler: - enabled: false - - # Recommender responsible for generating recommendation for the object. - # List should be empty (then the default recommender will generate the recommendation) - # or contain exactly one recommender. - # recommenders: - # - name: custom-recommender-performance - - # List of resources that the vertical pod autoscaler can control. Defaults to cpu and memory - controlledResources: [] - # Specifies which resource values should be controlled: RequestsOnly or RequestsAndLimits. - # controlledValues: RequestsAndLimits - - # Define the max allowed resources for the pod - maxAllowed: {} - # cpu: 200m - # memory: 100Mi - # Define the min allowed resources for the pod - minAllowed: {} - # cpu: 200m - # memory: 100Mi - - # updatePolicy: - # Specifies minimal number of replicas which need to be alive for VPA Updater to attempt pod eviction - # minReplicas: 1 - # Specifies whether recommended updates are applied when a Pod is started and whether recommended updates - # are applied during the life of a Pod. Possible values are "Off", "Initial", "Recreate", and "Auto". - # updateMode: Auto - -# Extra manifests to deploy as an array -extraManifests: [] - # - | - # apiVersion: v1 - # kind: ConfigMap - # metadata: - # name: prometheus-extra - # data: - # extra-data: "value" diff --git a/helm-overrides/k8s-supply-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml deleted file mode 100644 index 8125d5c..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/prometheus-stackdriver-exporter/custom-values.yaml +++ /dev/null @@ -1,170 +0,0 @@ -# Provide a name in place of prometheus-stackdriver-exporter for `app:` labels -nameOverride: "stackdriver-exporter-supply-ase1c-prd" - -# Provide a name to substitute for the full names of resources -fullnameOverride: "stackdriver-exporter-supply-ase1c-prd" - -# Number of exporters to run -replicaCount: 1 - -# Restart policy for container -restartPolicy: Always - -image: - repository: prometheuscommunity/stackdriver-exporter - # if not set appVersion field from Chart.yaml is used - tag: "" - pullPolicy: IfNotPresent - - ## Optionally specify an array of imagePullSecrets. - ## Secrets must be manually created in the namespace. - ## ref: https://kubernetes.io/docs/tasks/configure-pod-container/pull-image-private-registry/ - ## - # pullSecrets: - # - myDockerConfigJsonSecretName - -resources: - requests: - cpu: 500m - memory: 512Mi - # limits: - # cpu: 100m - # memory: 128Mi - -securityContext: {} - -containerSecurityContext: {} - -service: - type: ClusterIP - httpPort: 9255 - annotations: {} - -## Additional labels to add to all resources -customLabels: - bu: "supply" - team: "supply-sre" - service: "stackdriver-exporter-supply-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - zone_extended: "ase1c" - # app: prometheus-stackdriver-exporter - -secret: - labels: {} - -stackdriver: - # The Google Project ID to gather metrics for - projectId: "meesho-supply-ase1c-prd-0225" - # An existing secret which contains credentials.json - serviceAccountSecret: "" - # Provide custom key for the existing secret to load credentials.json from - serviceAccountSecretKey: "" - # A service account key JSON file. Must be provided when no existing secret is used, in this case a new secret will be created holding this service account - serviceAccountKey: "" - # Max number of retries that should be attempted on 503 errors from Stackdriver - maxRetries: 0 - # How long should Stackdriver_exporter wait for a result from the Stackdriver API - httpTimeout: 10s - # Max time between each request in an exp backoff scenario - maxBackoff: 5s - # The amount of jitter to introduce in an exp backoff scenario - backoffJitter: 1s - # The HTTP statuses that should trigger a retry - retryStatuses: 503 - # Drop metrics from attached projects and fetch `project_id` only - dropDelegatedProjects: false - metrics: - # The prefixes to gather metrics for, we default to just CPU metrics. - typePrefixes: 'bigtable.googleapis.com/client,bigtable.googleapis.com/cluster,bigtable.googleapis.com/disk,bigtable.googleapis.com/replication,bigtable.googleapis.com/server,bigtable.googleapis.com/table,cloudsql.googleapis.com/database,compute.googleapis.com/instance,logging.googleapis.com/user/node_drain_metric,compute.googleapis.com/guest/system/uptime' - # The filters to refine the metrics query by using Filter objects that Google provides. - # Filter objects: project, group.id, resource.type, resource.labels.[KEY], metric.type, metric.labels.[KEY] - # https://cloud.google.com/monitoring/api/v3/filters - filters: [] - # - 'pubsub.googleapis.com/subscription:resource.labels.subscription_id=monitoring.regex.full_match("us-west4.*my-team.*")' - # The frequency to request - interval: '5m' - # How far into the past to offset - offset: '0s' - # Offset for the Google Stackdriver Monitoring Metrics interval into the past by the ingest delay from the metric's metadata. - ingestDelay: false - # If enabled will treat all DELTA metrics as an in-memory counter instead of a gauge. - aggregateDeltas: false - # How long should a delta metric continue to be exported after GCP stops producing a metric - aggregateDeltasTTL: '30m' - -web: - # Port to listen on - listenAddress: ':9255' - # Path under which to expose metrics. - path: /metrics - -## Pod affinity -## -affinity: {} - -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "9255" - prometheus.io/path: "/metrics" - -## Pod extra arguments -## -extraArgs: {} - -## Node labels for stackdriver-exporter pod assignment -## Ref: https://kubernetes.io/docs/user-guide/node-selection/ -## -nodeSelector: - dedicated: "vmselect" - -## Node tolerations for stackdriver-exporter scheduling to nodes with taints -## Ref: https://kubernetes.io/docs/concepts/configuration/assign-pod-node/ -## -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - -## Service Account -## -serviceAccount: - # Specifies whether a ServiceAccount should be created - create: true - # The name of the ServiceAccount to use. - # If not set and create is false, 'default' is used - # If not set and create is true, a name is generated using the fullname template - name: - annotations: { - iam.gke.io/gcp-service-account: sa-supl-cnsre-stackdriver-prd@meesho-supply-ase1c-prd-0622.iam.gserviceaccount.com - } - -# Enable this if you're using https://github.com/coreos/prometheus-operator -serviceMonitor: - enabled: false - namespace: monitoring - # additionalLabels is the set of additional labels to add to the ServiceMonitor - additionalLabels: {} - # How long until a scrape request times out. - scrapeTimeout: '10s' - # fallback to the prometheus default unless specified - interval: 10s - # Defaults to what's used if you follow CoreOS [Prometheus Install Instructions](https://github.com/helm/charts/tree/master/stable/prometheus-operator#tldr) - honorLabels: true - # Whether Prometheus should use the timestamps of the metrics exposed by stackdriver-exporter - honorTimestamps: true - # MetricRelabelConfigs to apply to samples before ingestion https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - metricRelabelings: [] - # RelabelConfigs to apply to samples before scraping. https://github.com/prometheus-operator/prometheus-operator/blob/master/Documentation/api.md#relabelconfig - relabelings: [] - -## Custom PrometheusRules to be defined -## ref: https://github.com/coreos/prometheus-operator#customresourcedefinitions -prometheusRule: - enabled: false - additionalLabels: {} - namespace: "" - rules: [] diff --git a/helm-overrides/k8s-supply-prd-ase1c/telegraf-operator/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/telegraf-operator/custom-values.yaml deleted file mode 100644 index f3b41d8..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/telegraf-operator/custom-values.yaml +++ /dev/null @@ -1,151 +0,0 @@ -replicaCount: 3 -image: - repository: quay.io/influxdb/telegraf-operator - pullPolicy: IfNotPresent - sidecarImage: "asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/telegraf:1.24.4" - -dedicatedValue: false -schedulerName: default-scheduler - -classes: - secretName: "telegraf-operator-classes" - default: "infra" - data: - infra: | - [[inputs.mem]] - [[outputs.file]] - files = ["stdout"] - [[outputs.prometheus_client]] - listen = ":9273" - metric_version = 2 - path = "/metrics" - expiration_interval = "60s" - export_timestamp = false - [agent] - interval = "10s" - round_interval = true - metric_batch_size = 1000 - metric_buffer_limit = 40000 - collection_jitter = "0s" - flush_interval = "30s" - flush_jitter = "0s" - precision = "" - debug = true - hostname = "" - omit_hostname = true - [[aggregators.basicstats]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - stats = ["count", "min", "max", "mean", "sum"] - [[aggregators.histogram]] - period = "60s" - grace = "10s" - delay = "30s" - drop_original = false - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "CONTROLLER" - [[aggregators.histogram.config]] - buckets = [5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "RDS" - [[aggregators.histogram.config]] - buckets = [1.0, 2.0, 3.0, 4.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0, 1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "REDIS" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HBASE" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "HTTP" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "METHOD" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "QUERY_EXECUTION_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "DOWN_STREAM_LATENCY" - [[aggregators.histogram.config]] - buckets = [0.0, 5.0, 10.0, 20.0, 30.0, 40.0, 50.0, 60.0, 80.0, 100.0, 120.0, 150.0, 175.0, 200.0, 225.0, 250.0, 275.0, 300.0, 350.0, 400.0, 450.0, 500.0, 600.0, 700.0, 800.0, 900.0,1000.0, 1200.0, 1500.0, 2000.0, 5000.0, 10000.0, 12500.0, 15000.0, 20000.0, 25000.0, 30000.0] - measurement_name = "API_LATENCY" - - [[inputs.socket_listener]] - service_address = "udp://:8094" - read_buffer_size = "16MB" - [[inputs.statsd]] - protocol = "udp" - service_address = ":8125" - delete_gauges = false - delete_counters = false - percentiles = [50.0, 90.0, 99.0, 99.9, 99.95, 100.0] - datadog_extensions = true - allowed_pending_messages = 100000 - -certManager: - enable: false - -imagePullSecrets: [] -nameOverride: "" -fullnameOverride: "" -serviceAccount: - # Annotations to add to the service account - annotations: {} -podSecurityContext: {} - # fsGroup: 2000 -securityContext: {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 -resources: - limits: - cpu: 200m - memory: 256Mi - requests: - cpu: 50m - memory: 64Mi -sidecarResources: - requests: - cpu: 200m - memory: 200Mi -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: exporter - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: exporter -labels: - bu: "supply" - team: "supply-sre" - service: "telegraf-operator-supply-ase1c-prd" - env: "prd" - priority: "p0" - type: "exporter" - zone_extended: "ase1c" - -nodeSelector: - dedicated: "vmselect" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - -affinity: {} -requireAnnotationsForSecret: false -# allow hot reload ; disabled by default to support versions of telegraf -# that do not support hot-reload and --watch-config flag -hotReload: false diff --git a/helm-overrides/k8s-supply-prd-ase1c/victoria-metrics-agent/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/victoria-metrics-agent/custom-values.yaml deleted file mode 100644 index d7e27cc..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/victoria-metrics-agent/custom-values.yaml +++ /dev/null @@ -1,298 +0,0 @@ -# Default values for victoria-metrics-agent. -# This is a YAML-formatted file. -# Declare variables to be passed into your templates. - -replicaCount: 2 - -fullnameOverride: vmagent-supply-ase1c-prd -# vmagent scraping configuration: -# https://github.com/VictoriaMetrics/VictoriaMetrics/blob/master/docs/vmagent.md#how-to-collect-metrics-in-prometheus-format - -# use existing configmap if specified -# otherwise .config values will be used -configMap: "vmagent-supply-ase1c-prd-config" # Use same name as in fullnameOverride-config - -dedicatedValue: false - - -topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmagent - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmagent - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/ -deployment: - enabled: true - - # vmagent pods will take almost 20-25 mins to work properly - minReadySeconds: 180 - progressDeadlineSeconds: 300 - - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - -# ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/ -statefulset: - enabled: false - # -- create cluster of vmagents. See https://docs.victoriametrics.com/vmagent.html#scraping-big-number-of-targets - # available since 1.77.2 version https://github.com/VictoriaMetrics/VictoriaMetrics/releases/tag/v1.77.2 - clusterMode: false - # -- replication factor for vmagent in cluster mode - replicationFactor: 1 - # ref: https://kubernetes.io/docs/concepts/workloads/controllers/statefulset/#update-strategies - updateStrategy: {} - # type: RollingUpdate - - -image: - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmagent - tag: v1.93.7-cluster # rewrites Chart.AppVersion - pullPolicy: IfNotPresent - -imagePullSecrets: [] -nameOverride: "" - -containerWorkingDir: "/" - -rbac: - create: true - # Note: The PSP will only be deployed, if Kubernetes (<1.25) supports the resource. - pspEnabled: true - annotations: {} - extraLabels: {} - # -- if true and `rbac.enabled`, will deploy a Role/Rolebinding instead of a ClusterRole/ClusterRoleBinding - namespaced: false - -serviceAccount: - # Specifies whether a service account should be created - create: true - # Annotations to add to the service account - annotations: { - iam.gke.io/gcp-service-account: sa-common-np-supl-prd@meesho-supply-ase1c-prd-0625.iam.gserviceaccount.com - } - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: - -## See `kubectl explain poddisruptionbudget.spec` for more -## ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ -podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - -# WARN: need to specify at least one remote write url or one multi tenant url -# remoteWriteUrls: [] -remoteWriteUrls: - # - http://vminsert-supply-ase1c-prd.supl-c.meeshogcp.in/insert/100/prometheus/api/v1/write - - http://vminsert-supply-ase1c-prd.victoriametrics.svc.cluster.local:8480/insert/multitenant/prometheus/api/v1/write - # - http://vminsert-supply-ase1c-prd.victoriametrics.svc.cluster.local:8480/insert/100/prometheus/api/v1/write -# - http://prometheus:8480/insert/0/prometheus - -multiTenantUrls: [] -# multiTenantUrls: -# - http://vm-insert-az1:8480 -# - http://vm-insert-az2:8480 - -extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - promscrape.config.strictParse: false - promscrape.maxScrapeSize: 1000000000 - promscrape.minResponseSizeForStreamParse: 1000000 - loggerTimezone: "Asia/Kolkata" - - # Uncomment and specify the port if you want to support any of the protocols: - # https://victoriametrics.github.io/vmagent.html#features - # graphiteListenAddr: ":2003" - # influxListenAddr: ":8189" - # opentsdbHTTPListenAddr: ":4242" - # opentsdbListenAddr: ":4242" - -# -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables -env: - [] - # - name: VM_remoteWrite_basicAuth_password - # valueFrom: - # secretKeyRef: - # name: auth_secret - # key: password - -# extra Labels for Pods, Deployment and Statefulset -extraLabels: - bu: "supply" - team: "supply-sre" - service: "vmagent-supply-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmagent" - zone_extended: "ase1c" - - - -# extra Labels for Pods only -podLabels: {} - -# Additional hostPath mounts -extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - -# Extra Volumes for the pod -extraVolumes: - [] - # - name: example - # configMap: - # name: example - -# Extra Volume Mounts for the container -extraVolumeMounts: - [] - # - name: example - # mountPath: /example - -extraContainers: [] -# - name: config-reloader -# image: reloader-image - -podSecurityContext: - {} - # fsGroup: 2000 - -securityContext: - {} - # capabilities: - # drop: - # - ALL - # readOnlyRootFilesystem: true - # runAsNonRoot: true - # runAsUser: 1000 - -service: - enabled: true - annotations: {} - # cloud.google.com/neg: '{"exposed_ports": {"8429":{"name": "vmagent-supply-prd"}}}' - extraLabels: {} - clusterIP: "" - ## Ref: https://kubernetes.io/docs/user-guide/services/#external-ips - ## - externalIPs: [] - loadBalancerIP: "" - loadBalancerSourceRanges: [] - servicePort: 8429 - # nodePort: 30000 - type: ClusterIP - # Ref: https://kubernetes.io/docs/tasks/access-application-cluster/create-external-load-balancer/#preserving-the-client-source-ip - # externalTrafficPolicy: "local" - # healthCheckNodePort: 0 -ingress: - enabled: true - ingressClassName: nginx-internal - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/rewrite-target: / - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - - extraLabels: {} - hosts: - - name: vmagent-supply-ase1c-prd.meeshogcp.in - path: / - port: http - tls: [] - # - secretName: vmagent-ingress-tls - # hosts: - # - vmagent.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - -resources: - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - requests: - cpu: 12 - memory: 10Gi - -# Annotations to be added to the deployment -annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -# Annotations to be added to pod -podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8429" - -nodeSelector: - dedicated: "vmagent" - -tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmagent" - effect: "NoSchedule" - - -affinity: {} - -# -- priority class to be assigned to the pod(s) -priorityClassName: "" - -serviceMonitor: - enabled: false - extraLabels: {} - annotations: {} - relabelings: [] -# interval: 15s -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - -persistence: - enabled: false - # storageClassName: default - accessModes: - - ReadWriteOnce - size: 10Gi - annotations: {} - extraLabels: {} - existingClaim: "" - # -- Bind Persistent Volume by labels. Must match all labels of targeted PV. - matchLabels: {} - -# -- Extra scrape configs that will be appended to `config` -extraScrapeConfigs: [] - -# Add extra specs dynamically to this chart -extraObjects: [] diff --git a/helm-overrides/k8s-supply-prd-ase1c/victoria-metrics-insert/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/victoria-metrics-insert/custom-values.yaml deleted file mode 100644 index 443f978..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/victoria-metrics-insert/custom-values.yaml +++ /dev/null @@ -1,235 +0,0 @@ -vmselect: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-supply-ase1c-prd - replicaCount: 5 - -dedicatedValue: false -# schedulerName: default-scheduler - -vminsert: - # -- Enable deployment of vminsert component. Deployment is used - enabled: true - # -- vminsert container name - name: vminsert - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vminsert - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vminsert component - fullnameOverride: vminsert-supply-ase1c-prd - # Extra command line arguments for vminsert component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - maxLabelsPerTimeseries: 40 - replicationFactor: 1 - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - extraLabels: - bu: "supply" - team: "supply-sre" - service: "vminsert-supply-ase1c-prd" - env: "prd" - priority: "p0" - type: "vminsert" - zone_extended: "ase1c" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - -# Horizontal Pod Autoscaling - horizontalPodAutoscaler: - # -- Use HPA for vminsert component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vminsert component - maxReplicas: 10 - # -- Minimum replicas for HPA to use to scale the vminsert component - minReplicas: 1 - # -- Metric for HPA to use to scale the vminsert component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vminsert" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vminsert" - - # -- Pod affinity - affinity: {} - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vminsert - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vminsert - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8480" - # -- Count of vminsert pods - replicaCount: 1 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 5 - memory: 2Gi - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips]( https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balancer IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8480 - # -- Target port - targetPort: http - # -- Service type - type: NodePort - # -- Enable UDP port. used if you have "spec.opentsdbListenAddr" specified - # -- Make sure that service is not type "LoadBalancer", as it requires "MixedProtocolLBService" feature gate. ref: https://kubernetes.io/docs/reference/command-line-tools-reference/feature-gates/ - udp: false - ingress: - # -- Enable deployment of ingress for vminsert component - enabled: false - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - extraLabels: {} - # -- Array of host objects - hosts: - - name: vminsert-supply-ase1c-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vminsert-ingress-tls - # hosts: - # - vminsert.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - ingressClassName: internal - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - serviceMonitor: - # -- Enable deployment of Service Monitor for vminsert component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vminsert component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vminsert component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-supply-prd-ase1c/victoria-metrics-select/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/victoria-metrics-select/custom-values.yaml deleted file mode 100644 index 130b0a6..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/victoria-metrics-select/custom-values.yaml +++ /dev/null @@ -1,301 +0,0 @@ -vminsert: - enabled: false - -vmstorage: - enabled: false - fullnameOverride: vmstorage-supply-ase1c-prd - replicaCount: 5 - -dedicatedValue: false - - -vmselect: - # -- Enable deployment of vmselect component. Can be deployed as Deployment(default) or StatefulSet - enabled: true - # -- Vmselect container name - name: vmselect - strategy: {} - # rollingUpdate: - # maxSurge: 25% - # maxUnavailable: 25% - # type: RollingUpdate - #extraVMSelects: [] - - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmselect - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmselect component - fullnameOverride: vmselect-supply-prd - # -- Suppress rendering `--storageNode` FQDNs based on `vmstorage.replicaCount` value. If true suppress rendering `--storageNodes`, they can be re-defined in extraArgs - suppresStorageFQDNsRender: false - automountServiceAccountToken: true - # Extra command line arguments for vmselect component - splitService: True - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - clusternativeListenAddr: ":8401" - clusternative.maxConcurrentRequests: "384" - dedup.minScrapeInterval: 60s - search.maxConcurrentRequests: "384" - search.maxSamplesPerQuery: "1000000000000" - search.maxQueryDuration: 300s - search.maxQueueDuration: 60s - search.maxSeries: "10000000000" - search.maxExportSeries: "1000000000" - search.maxUniqueTimeseries: "1000000000000" - search.maxQueryLen: "1000000" - loggerTimezone: "Asia/Kolkata" - - annotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - extraLabels: - bu: "supply" - team: "supply-sre" - service: "vmselect-supply-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmselect" - zone_extended: "ase1c" - # arch: "arm64" - # runpod: "ondemand" - - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - - # Readiness & Liveness probes - probe: - readiness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - - horizontalPodAutoscaler: - # -- Use HPA for vmselect component - enabled: true - # -- Maximum replicas for HPA to use to to scale the vmselect component - maxReplicas: 15 - # -- Minimum replicas for HPA to use to scale the vmselect component - minReplicas: 1 - # -- Metric for HPA to use to scale the vmselect component - metrics: - - type: Resource - resource: - name: cpu - target: - type: Utilization - averageUtilization: 40 - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - initContainers: - [] - # - name: example - # image: example-image - - podDisruptionBudget: - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: https://kubernetes.io/docs/tasks/run-application/configure-pdb/ - enabled: true - minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - - nodeSelector: - dedicated: "vmselect" - - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmselect" - effect: "NoSchedule" - - - # -- Pod affinity - affinity: {} - # -- Pod topologySpreadConstraints - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmselect - # -- Pod's annotations - podAnnotations: - prometheus.io/scrape: "true" - prometheus.io/port: "8481" - # -- Count of vmselect pods - replicaCount: 2 - # -- Container workdir - containerWorkingDir: "" - # -- Resource object - resources: - # limits: - # cpu: 50m - # memory: 64Mi - requests: - cpu: 10 - memory: 32Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/ - securityContext: {} - podSecurityContext: {} - # -- Cache root folder - cacheMountPath: /cache - service: - # -- Service annotations - annotations: - io.cilium/global-service: "true" - # -- Service labels - labels: {} - # -- Service ClusterIP - clusterIP: "" - # -- Service External IPs. Ref: [https://kubernetes.io/docs/user-guide/services/#external-ips](https://kubernetes.io/docs/user-guide/services/#external-ips) - externalIPs: [] - # -- Extra service ports - extraServicePorts: [] - # -- Service load balacner IP - loadBalancerIP: "" - # -- Load balancer source range - loadBalancerSourceRanges: [] - # -- Service port - servicePort: 8481 - # -- Target port - targetPort: http - # -- Service type - type: ClusterIP - ingress: - # -- Enable deployment of ingress for vmselect component - enabled: true - # -- Ingress annotations - annotations: - nginx.ingress.kubernetes.io/force-ssl-redirect: "false" - nginx.ingress.kubernetes.io/ssl-redirect: "false" - # kubernetes.io/ingress.class: nginx - # kubernetes.io/tls-acme: 'true' - ingressClassName: nginx-internal - - extraLabels: {} - # -- Array of host objects - hosts: - - name: vmselect-supply-ase1c-prd.meeshogcp.in - path: / - port: http - # -- Array of TLS objects - tls: [] - # - secretName: vmselect-ingress-tls - # hosts: - # - vmselect.local - # For Kubernetes >= 1.18 you should specify the ingress-controller via the field ingressClassName - # See https://kubernetes.io/blog/2020/04/02/improvements-to-the-ingress-api-in-kubernetes-1.18/#specifying-the-class-of-an-ingress - # ingressClassName: nginx - # -- pathType is only for k8s >= 1.1= - pathType: Prefix - - clusternativeService: - enabled: true - port: 8401 - sessionAffinity: None - ipFamilyPolicy: SingleStack - - statefulSet: - # -- Deploy StatefulSet instead of Deployment for vmselect. Useful if you want to keep cache data. - enabled: false - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - ## Headless service for statefulset - service: - # -- Headless service annotations - annotations: {} - # -- Headless service labels - labels: {} - # -- Headless service port - servicePort: 8481 - persistentVolume: - # -- Create/use Persistent Volume Claim for vmselect component. Empty dir if false. If true, vmselect will create/use a Persistent Volume Claim - enabled: false - - # -- Array of access mode. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - - # -- Persistent volume labels - labels: {} - - # -- Existing Claim name. Requires vmselect.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - ## Vmselect data Persistent Volume mount root path - ## - # -- Size of the volume. Better to set the same as resource limit memory property - size: 2Gi - # -- Mount subpath - subPath: "" - serviceMonitor: - # -- Enable deployment of Service Monitor for vmselect component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmselect component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmselect component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/helm-overrides/k8s-supply-prd-ase1c/victoria-metrics-storage/custom-values.yaml b/helm-overrides/k8s-supply-prd-ase1c/victoria-metrics-storage/custom-values.yaml deleted file mode 100644 index fce5852..0000000 --- a/helm-overrides/k8s-supply-prd-ase1c/victoria-metrics-storage/custom-values.yaml +++ /dev/null @@ -1,317 +0,0 @@ -vminsert: - enabled: false - -vmselect: - enabled: false - -serviceAccount: - create: true - # name: - extraLabels: {} - annotations: - eks.amazonaws.com/role-arn: arn:aws:iam::847438129436:role/eks-s3-vmbackup-role - # mount API token to pod directly - automountToken: true - -dedicatedValue: false -# schedulerName: default-scheduler - -vmstorage: - # -- Enable deployment of vmstorage component. StatefulSet is used - enabled: true - # -- vmstorage container name - name: vmstorage - image: - # -- Image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmstorage - # -- Image tag - tag: v1.93.7-cluster - # -- Image pull policy - pullPolicy: IfNotPresent - # -- Name of Priority Class - priorityClassName: "" - # -- Overrides the full name of vmstorage component - fullnameOverride: vmstorage-supply-ase1c-prd - automountServiceAccountToken: true - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - # -- Data retention period. Supported values 1w, 1d, number without measurement means month, e.g. 2 = 2month - retentionPeriod: 90d - # Additional vmstorage container arguments. Extra command line arguments for vmstorage component - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - search.maxUniqueTimeseries: "30000000" - dedup.minScrapeInterval: 60s - loggerTimezone: "Asia/Kolkata" - - # Additional hostPath mounts - extraHostPathMounts: - [] - # - name: certs-dir - # mountPath: /etc/kubernetes/certs - # subPath: "" - # hostPath: /etc/kubernetes/certs - # readOnly: true - - # Extra Volumes for the pod - extraVolumes: - [] - # - name: example - # configMap: - # name: example - - # Extra Volume Mounts for the container - extraVolumeMounts: - [] - # - name: example - # mountPath: /example - - extraContainers: - [] - # - name: config-reloader - # image: reloader-image - - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - - initContainers: - [] - # - name: vmrestore - # image: victoriametrics/vmrestore:latest - # volumeMounts: - # - mountPath: /storage - # name: vmstorage-volume - # - mountPath: /etc/vm/creds - # name: secret-remote-storage-keys - # readOnly: true - # args: - # - -storageDataPath=/storage - # - -src=s3://your_bucket/folder/latest - # - -credsFilePath=/etc/vm/creds/credentials - - # -- See `kubectl explain poddisruptionbudget.spec` for more. Ref: [https://kubernetes.io/docs/tasks/run-application/configure-pdb/](https://kubernetes.io/docs/tasks/run-application/configure-pdb/) - podDisruptionBudget: - enabled: false - # minAvailable: 1 - # maxUnavailable: 1 - labels: {} - - # -- Array of tolerations object. Node tolerations for server scheduling to nodes with taints. Ref: [https://kubernetes.io/docs/concepts/configuration/assign-pod-node/](https://kubernetes.io/docs/concepts/configuration/assign-pod-node/) - ## - tolerations: - - key: "dedicated" - operator: "Equal" - value: "vmstorage" - effect: "NoSchedule" - - # -- Pod's node selector. Ref: [https://kubernetes.io/docs/user-guide/node-selection/](https://kubernetes.io/docs/user-guide/node-selection/) - nodeSelector: - dedicated: "vmstorage" - - # -- Pod affinity - affinity: {} - - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: topology.kubernetes.io/zone - whenUnsatisfiable: ScheduleAnyway - labelSelector: - matchLabels: - type: vmstorage - - maxSkew: 1 - topologyKey: kubernetes.io/hostname - whenUnsatisfiable: DoNotSchedule - labelSelector: - matchLabels: - type: vmstorage - - ## Use an alternate scheduler, e.g. "stork". - ## ref: https://kubernetes.io/docs/tasks/administer-cluster/configure-multiple-schedulers/ - ## - # schedulerName: - - persistentVolume: - # -- Create/use Persistent Volume Claim for vmstorage component. Empty dir if false. If true, vmstorage will create/use a Persistent Volume Claim - enabled: true - - # -- Array of access modes. Must match those of existing PV or dynamic provisioner. Ref: [http://kubernetes.io/docs/user-guide/persistent-volumes/](http://kubernetes.io/docs/user-guide/persistent-volumes/) - accessModes: - - ReadWriteOnce - # -- Persistent volume annotations - annotations: {} - # -- Persistent volume labels - labels: {} - # -- Storage class name. Will be empty if not setted - storageClass: pd-standard-retain - # -- Existing Claim name. Requires vmstorage.persistentVolume.enabled: true. If defined, PVC must be created manually before volume will be bound - existingClaim: "" - - # -- Data root path. Vmstorage data Persistent Volume mount root path - mountPath: /storage - # -- Size of the volume. Better to set the same as resource limit memory property - size: 250Gi - # -- Mount subpath - subPath: "" - - # -- Pod's annotations - podAnnotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - annotations: - prometheus.io/port: "8482" - prometheus.io/scrape: "true" - extraLabels: - bu: "supply" - team: "supply-sre" - service: "vmstorage-supply-ase1c-prd" - env: "prd" - priority: "p0" - type: "vmstorage" - zone_extended: "ase1c" - # arch: "arm64" - # runpod: "ondemand" - - # -- Count of vmstorage pods - replicaCount: 5 - # -- Container workdir - containerWorkingDir: "" - # -- Deploy order policy for StatefulSet pods - podManagementPolicy: OrderedReady - - # -- Resource object. Ref: [https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/) - resources: - # limits: - # cpu: 500m - # memory: 512Mi - requests: - cpu: 12 - memory: 160Gi - - # -- Pod's security context. Ref: [https://kubernetes.io/docs/tasks/configure-pod-container/security-context/](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/) - securityContext: {} - podSecurityContext: {} - service: - # -- Service annotations - annotations: {} - # -- Service labels - labels: {} - # -- Service port - servicePort: 8482 - # -- Port for accepting connections from vminsert - vminsertPort: 8400 - # -- Port for accepting connections from vmstorage - vmstoragePort: 8401 - # -- Extra service ports - extraServicePorts: [] - # -- Pod's termination grace period in seconds - terminationGracePeriodSeconds: 60 - probe: - readiness: - httpGet: - path: /health - port: http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - liveness: - tcpSocket: - port: http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - - vmbackupmanager: - # -- enable automatic creation of backup via vmbackupmanager. vmbackupmanager is part of Enterprise packages - enable: false - # -- should be true and means that you have the legal right to run a backup manager - # that can either be a signed contract or an email with confirmation to run the service in a trial period - # # https://victoriametrics.com/legal/eula/ - eula: true - image: - # -- vmbackupmanager image repository - repository: asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/vmbackupmanager - # -- vmbackupmanager image tag - tag: v1.93.7-cluster - # -- disable hourly backups - disableHourly: true - # -- disable daily backups - disableDaily: false - # -- disable weekly backups - disableWeekly: true - # -- disable monthly backups - disableMonthly: true - # -- backup destination at S3, GCS or local filesystem. Pod name will be included to path! - destination: "s3://prd-supply-ase1c-vmbackup" - # -- backups' retention settings - retention: - # -- keep last N hourly backups. 0 means delete all existing hourly backups. Specify -1 to turn off - keepLastHourly: 0 - # -- keep last N daily backups. 0 means delete all existing daily backups. Specify -1 to turn off - keepLastDaily: 30 - # -- keep last N weekly backups. 0 means delete all existing weekly backups. Specify -1 to turn off - keepLastWeekly: 0 - # -- keep last N monthly backups. 0 means delete all existing monthly backups. Specify -1 to turn off - keepLastMonthly: 0 - extraArgs: - envflag.enable: "true" - envflag.prefix: VM_ - loggerFormat: json - concurrency: 15 - loggerTimezone: "Asia/Kolkata" - # -- Allows to enable restore options for pod. - # Read more: https://docs.victoriametrics.com/vmbackupmanager.html#restore-commands - restore: - onStart: - enabled: false - resources: {} - # -- Additional environment variables (ex.: secret tokens, flags) https://github.com/VictoriaMetrics/VictoriaMetrics#environment-variables - env: [] - readinessProbe: - httpGet: - path: /health - port: manager-http - initialDelaySeconds: 5 - periodSeconds: 15 - timeoutSeconds: 5 - failureThreshold: 3 - livenessProbe: - tcpSocket: - port: manager-http - initialDelaySeconds: 30 - periodSeconds: 30 - timeoutSeconds: 5 - failureThreshold: 10 - extraSecretMounts: - [] - # - name: secret - # mountPath: /etc/credentials - # subPath: "" - # readOnly: true - serviceMonitor: - # -- Enable deployment of Service Monitor for vmstorage component. This is Prometheus operator object - enabled: false - # -- Target namespace of ServiceMonitor manifest - namespace: "" - # -- Service Monitor labels - extraLabels: {} - # -- Service Monitor annotations - annotations: {} - # Commented. Prometheus scare interval for vmstorage component -# interval: 15s - # Commented. Prometheus pre-scrape timeout for vmstorage component -# scrapeTimeout: 5s - # -- Commented. HTTP scheme to use for scraping. -# scheme: https - # -- Commented. TLS configuration to use when scraping the endpoint -# tlsConfig: -# insecureSkipVerify: true - # -- Service Monitor relabelings - relabelings: [] diff --git a/index.md b/index.md deleted file mode 100644 index 8afca99..0000000 --- a/index.md +++ /dev/null @@ -1,123 +0,0 @@ -# Documentation Index — `devops-infra-helm-charts` - -> Repo: GitOps Helm values + cached/forked charts for Meesho's GKE infrastructure fleet. -> -> **Layer:** Layer 1 — Agent-Writable (config repo). -> -> Repo entry point for agents: [/CLAUDE.md](CLAUDE.md). - -This index mirrors the structure of the agent-documentation pyramid: governance at the top, schemas underneath, procedures and runbooks above the operational floor, skills wrapping procedures for agents, ADRs explaining the why. - ---- - -## Layout - -```text -/ -├── CLAUDE.md Agent entry point (Layer + naming + layout + layer table) -├── README.md Short repo intro for humans -├── index.md ← you are here -├── repository.yaml Owner metadata (registry-bootstrap-managed) -│ -├── docs/ -│ ├── architecture.md Full deploy lifecycle, fleet, chart inventory, hooks -│ ├── global/ Cross-cutting governance -│ │ ├── AGENT_BOUNDARIES.md Layer 1 / 2 / 3 map for THIS repo, blast radii -│ │ ├── SANCTITY_RULES.md Numbered non-negotiable rules -│ │ └── coding-guidelines/ -│ │ └── helm-values.md Patterns for authoring custom-values.yaml -│ └── platform/ -│ ├── procedures/ "How to make change X" — six recipes -│ │ ├── onboard-app-to-cluster.md -│ │ ├── onboard-new-cluster.md -│ │ ├── update-chart-version.md -│ │ ├── fork-upstream-chart.md -│ │ ├── blue-green-chart-migration.md -│ │ └── deboard-app.md -│ ├── runbooks/ Symptom → diagnosis decision trees -│ │ ├── argocd-sync-failure.md -│ │ ├── ingress-down.md -│ │ └── pod-pending-scheduling.md -│ └── schemas/ Field-by-field YAML annotation -│ ├── custom-values-schema.md -│ ├── raw-manifest-sidecar-schema.md -│ └── storageclass-priorityclass-schema.md -│ -├── skills/ -│ └── infra/ Agent-callable parameterised tasks -│ ├── onboard-app.md -│ ├── bump-chart-version.md -│ └── diagnose-scheduling.md -│ -├── wiki/ -│ ├── entities/ -│ │ └── DevOps Infra Helm Charts.md Architectural reference entity for THIS repo -│ └── analyses/ Architecture Decision Records -│ ├── ADR-A1-cache-vs-upstream-charts.md -│ ├── ADR-A2-blue-green-sibling-pattern.md -│ ├── ADR-A3-per-cluster-scheduling.md -│ ├── ADR-A4-raw-manifest-sidecars-in-helm-overrides.md -│ └── ADR-A5-manual-sync-default-for-infra.md -│ -├── contour-nodeselector-tolerations-summary.md -│ Per-cluster Contour scheduling matrix -│ (load-bearing — read before any Contour edit) -│ -├── helm-templates// 74 cached/forked upstream charts -├── helm-overrides/// Cluster × app override values + sidecar manifests -├── manifests/ Cluster-wide singletons (storageclass / priorityclass / Jenkins / JFrog) -├── pre-commit-scripts/ TruffleHog (active); CAC + Yaak (no-op here) -└── post-commit-scripts/ Cursor metric collector (background, non-blocking) -``` - ---- - -## Quick navigation - -### "I'm an agent. What do I read first?" - -1. [/CLAUDE.md](CLAUDE.md) — entry point. Repo role, layer, NEVER DO list, layout, layer constraint table. -2. [docs/global/AGENT_BOUNDARIES.md](docs/global/AGENT_BOUNDARIES.md) — per-operation Layer 1 / 2 / 3 classification with blast radii. -3. [docs/global/SANCTITY_RULES.md](docs/global/SANCTITY_RULES.md) — numbered hard stops. -4. The relevant procedure or runbook for the task. - -### "I'm a human reviewing an agent-generated PR. What do I check?" - -1. [docs/platform/schemas/custom-values-schema.md](docs/platform/schemas/custom-values-schema.md) — does each field follow convention? -2. [docs/global/SANCTITY_RULES.md](docs/global/SANCTITY_RULES.md) — does the PR cross any non-negotiable? -3. The procedure used (if any) — did the PR follow it end-to-end? -4. [contour-nodeselector-tolerations-summary.md](contour-nodeselector-tolerations-summary.md) — if Contour is touched, do `nodeSelector`/`tolerations` match the cluster's row? - -### "Why is this repo shaped this way?" - -Read in order: -- [docs/architecture.md](docs/architecture.md) — the system context, module boundaries, deploy lifecycle. -- ADRs in [wiki/analyses/](wiki/analyses/) — five recorded decisions. - -### "Help me do task X." - -| Task | Doc | -|------|-----| -| Add a new app override to a cluster | [docs/platform/procedures/onboard-app-to-cluster.md](docs/platform/procedures/onboard-app-to-cluster.md) | -| Bring up a brand-new cluster's overrides | [docs/platform/procedures/onboard-new-cluster.md](docs/platform/procedures/onboard-new-cluster.md) | -| Bump a chart's pinned dependency version | [docs/platform/procedures/update-chart-version.md](docs/platform/procedures/update-chart-version.md) | -| Intentionally fork an upstream chart | [docs/platform/procedures/fork-upstream-chart.md](docs/platform/procedures/fork-upstream-chart.md) | -| Migrate a chart blue-green (sibling pattern) | [docs/platform/procedures/blue-green-chart-migration.md](docs/platform/procedures/blue-green-chart-migration.md) | -| Remove a retired app override | [docs/platform/procedures/deboard-app.md](docs/platform/procedures/deboard-app.md) | -| Argo CD app errored / OutOfSync | [docs/platform/runbooks/argocd-sync-failure.md](docs/platform/runbooks/argocd-sync-failure.md) | -| Ingress (Contour) is down on a cluster | [docs/platform/runbooks/ingress-down.md](docs/platform/runbooks/ingress-down.md) | -| Pods pending / wrong-node scheduling | [docs/platform/runbooks/pod-pending-scheduling.md](docs/platform/runbooks/pod-pending-scheduling.md) | - ---- - -## Repo facts (quick reference) - -| | | -|---|---| -| Charts in `helm-templates/` | 74 | -| Cluster directories under `helm-overrides/` | 30+ (`k8s-*-prd-ase1[c]`, `k8s-shared-int-ase1`, `k8s-aurva-prd-ase1`, `k8s-supply-dev-ase1`, plus `db-*` dataplane) | -| Singletons under `manifests/` | StorageClasses (4), per-cluster PriorityClasses, Jenkins/JFrog filestore PV/PVCs | -| Active pre-commit hooks | TruffleHog (verified-secret scan; webhook to `observe.meeshogcp.in`) | -| Build / test / lint | None — declarative YAML only | -| Deploy mechanism | Argo CD reconcile from `main` | -| Owners | `siddharth.pal@meesho.com` (primary), `samarth.nag@meesho.com` (secondary) — per `repository.yaml` | diff --git a/manifests/jenkins-filestore-caching/dev/pv.yaml b/manifests/jenkins-filestore-caching/dev/pv.yaml deleted file mode 100644 index 8b178bb..0000000 --- a/manifests/jenkins-filestore-caching/dev/pv.yaml +++ /dev/null @@ -1,20 +0,0 @@ -apiVersion: v1 -kind: PersistentVolume -metadata: - name: pv-jenkins-agents-cache-dev - annotations: - pv.kubernetes.io/provisioned-by: filestore.csi.storage.gke.io -spec: - storageClassName: sc-filestore-standard - capacity: - storage: 2Ti - accessModes: - - ReadWriteMany - persistentVolumeReclaimPolicy: Retain - volumeMode: Filesystem - csi: - driver: filestore.csi.storage.gke.io - volumeHandle: "modeInstance/asia-southeast1-c/fs-infr-dvops-jenkins-agents-dev-ase1/fsn_infr_dev" - volumeAttributes: - ip: 10.22.228.202 - volume: fsn_infr_dev diff --git a/manifests/jenkins-filestore-caching/dev/pvc.yaml b/manifests/jenkins-filestore-caching/dev/pvc.yaml deleted file mode 100644 index 017ada3..0000000 --- a/manifests/jenkins-filestore-caching/dev/pvc.yaml +++ /dev/null @@ -1,13 +0,0 @@ -kind: PersistentVolumeClaim -apiVersion: v1 -metadata: - name: pvc-jenkins-agents-cache-dev - namespace: jenkins -spec: - accessModes: - - ReadWriteMany - storageClassName: sc-filestore-standard - volumeName: pv-jenkins-agents-cache-dev - resources: - requests: - storage: 2Ti diff --git a/manifests/jfrog-filestore-data/dev/pv.yaml b/manifests/jfrog-filestore-data/dev/pv.yaml deleted file mode 100644 index 51a8549..0000000 --- a/manifests/jfrog-filestore-data/dev/pv.yaml +++ /dev/null @@ -1,21 +0,0 @@ -apiVersion: v1 -kind: PersistentVolume -metadata: - name: pv-infr-dvops-jenkins-agents-dev - annotations: - pv.kubernetes.io/provisioned-by: filestore.csi.storage.gke.io -spec: - storageClassName: sc-filestore-standard - capacity: - storage: 1Ti - accessModes: - - ReadWriteMany - persistentVolumeReclaimPolicy: Retain - volumeMode: Filesystem - csi: - driver: filestore.csi.storage.gke.io - # Modify this to use the zone, filestore instance and share name. - volumeHandle: "modeInstance/asia-southeast1-c/fs-infr-dvops-jenkins-agents-dev-ase1/fsn_infr_dev" - volumeAttributes: - ip: 10.76.232.194 # Modify this to Pre-provisioned Filestore instance IP - volume: fsn_infr_dev # Modify this to Pre-provisioned Filestore instance share name diff --git a/manifests/jfrog-filestore-data/dev/pvc.yaml b/manifests/jfrog-filestore-data/dev/pvc.yaml deleted file mode 100644 index a1db023..0000000 --- a/manifests/jfrog-filestore-data/dev/pvc.yaml +++ /dev/null @@ -1,13 +0,0 @@ -kind: PersistentVolumeClaim -apiVersion: v1 -metadata: - name: pvc-infr-dvops-jenkins-agents-dev - namespace: jenkins-dev -spec: - accessModes: - - ReadWriteMany - storageClassName: sc-filestore-standard - volumeName: pv-infr-dvops-jenkins-agents-dev - resources: - requests: - storage: 1Ti diff --git a/manifests/nginx-static-content/central/nginx-static-content-manifest.yaml b/manifests/nginx-static-content/central/nginx-static-content-manifest.yaml deleted file mode 100644 index 6da878d..0000000 --- a/manifests/nginx-static-content/central/nginx-static-content-manifest.yaml +++ /dev/null @@ -1,213 +0,0 @@ -# Resources from namespace "nginx-static-content" — source cluster context: gke-central-prd-ase1a -# Recreate in a new cluster with: kubectl apply -f nginx-static-content-manifest.yaml -# NOTE: requires the Contour CRD (projectcontour.io/HTTPProxy) installed in the target cluster. ---- -apiVersion: v1 -kind: Namespace -metadata: - name: nginx-static-content ---- -apiVersion: v1 -data: - nginx.conf: | - events { - worker_connections 512; - } - http { - server { - listen 80; - - location /akamai/sureroute-test-object { - alias /usr/share/nginx/html/; - try_files $uri $uri/ /index.html; - autoindex on; - } - - location / { - return 404; - } - } - } -kind: ConfigMap -metadata: - annotations: {} - name: nginx-config - namespace: nginx-static-content ---- -apiVersion: v1 -data: - index.html: | - - - - - - Neque porro quisquam est qui dolorem ipsum quia dolor sit amet consectetur adipisci velit - - - -

Lorem Ipsum

- -

Ipsum Lorem

- -
- - Lorem ipsum dolor sit amet consectetur adipiscing elit Nullam massa enim tincidunt non hendrerit eget malesuada et nisi In hac habitasse platea dictumst Praesent nec laoreet ante Aenean tempus nisi in erat tempus tempus Vestibulum imperdiet lobortis sapien eu tempus Vivamus volutpat quam sed eros molestie vitae dignissim nulla ultricies Vivamus dictum elit velit Pellentesque pellentesque ornare ornare Mauris vel gravida sapien Praesent eleifend tristique ipsum nec tempor Vestibulum cursus eleifend tellus a egestas lectus euismod sedDuis nec massa quam Nulla porta enim ut consequat tincidunt quam tortor consequat enim eu interdum eros lorem eu turpis Cras vestibulum orci quis felis tristique quis semper sem imperdiet Sed mattis tincidunt risus scelerisque scelerisque Aliquam nisl quam bibendum quis luctus eu sodales ut felis Integer id turpis nisi Phasellus mattis nulla eu odio faucibus a auctor orci tristique Nulla ullamcorper risus nec semper accumsan libero lacus aliquet elit quis lacinia metus nunc vestibulum turpis Suspendisse vel sapien vel magna auctor aliquam Aenean fringilla fringilla metus non imperdiet Aliquam nisl lacus tempus vitae commodo non accumsan ut lectus Nam in urna eu neque pretium aliquam Maecenas sit amet urna lectus Donec vitae metus enimSed lacus nulla faucibus eget ullamcorper ut mollis at metus Vivamus tortor felis tincidunt at tristique ut tincidunt feugiat velit Ut euismod felis non urna luctus luctus Integer nec urna massa Mauris vestibulum hendrerit auctor Morbi at tellus nec arcu scelerisque rhoncus Phasellus facilisis interdum lorem vulputate posuere Nullam quis felis est Aenean metus augue tempus non ultricies et dapibus vel felis Pellentesque at augue velit Nulla erat nisi posuere eu pellentesque id pretium ac libero Phasellus tincidunt sollicitudin sapien at mollis Nullam et libero velit nec tincidunt eros Aliquam et sem elit Quisque suscipit orci enim vel aliquam nisi Suspendisse in enim a ligula blandit volutpat in id velitNam tempor neque nec ligula sollicitudin rhoncus Etiam et lorem vel odio pharetra interdum Cum sociis natoque penatibus et magnis dis parturient montes nascetur ridiculus mus In imperdiet nisi sed diam rutrum gravida in vel massa Nam ullamcorper ultrices diam vitae consequat lacus consequat consequat Curabitur laoreet leo sed tortor fringilla nec euismod libero lobortis Donec non enim lectus Suspendisse potenti In hac habitasse platea dictumst Fusce semper auctor neque nec lobortis Praesent vitae mauris turpis Lorem ipsum dolor sit amet consectetur adipiscing elit Proin sed pharetra odio Suspendisse potenti Cum sociis natoque penatibus et magnis dis parturient montes nascetur ridiculus mus Duis eget odio purus quis dapibus massaCurabitur ut dapibus eros Donec tempor felis ac facilisis bibendum nisi purus pellentesque sem sollicitudin tempor lectus nulla at mi Maecenas quis urna ut ante pulvinar pellentesque Duis auctor imperdiet suscipit Pellentesque dui nulla volutpat quis posuere a gravida ornare augue Proin nec felis pharetra magna pellentesque facilisis Curabitur lacus libero malesuada sed tincidunt ac aliquet ut tortor Etiam gravida lorem nulla consectetur eleifend risus Donec facilisis turpis laoreet imperdiet laoreet purus justo egestas nulla et hendrerit leo eros at orci Nunc vulputate mauris sit amet sapien accumsan nec euismod orci volutpat Sed ultricies velit ut lorem venenatis in convallis tellus imperdiet Aenean auctor ultrices est ultricies rhoncus Phasellus non magna a leo luctus fermentum nec fermentum eratSed faucibus nisl quis diam mollis quis varius tortor tincidunt Phasellus in turpis in tellus consectetur mollis Donec a neque id metus condimentum dignissim In hac habitasse platea dictumst Pellentesque sem nisi pulvinar nec sagittis vitae lacinia non tellus Aliquam dignissim dignissim volutpat Pellentesque ut quam et mi tincidunt varius id vel quam Duis consectetur elit ac ligula fringilla elementum In elementum tellus viverra mi vehicula vitae tempus lectus laoreet Nullam diam nibh tincidunt vitae imperdiet a luctus a felis In posuere pulvinar volutpat Pellentesque eget viverra justoNullam nec sapien at felis molestie auctor Sed dignissim erat eu nulla ullamcorper mattis Curabitur felis sem feugiat non semper ut sollicitudin sed ipsum Quisque cursus laoreet turpis sit amet molestie neque consequat at Vestibulum eu ligula quis nisl pulvinar rhoncus Praesent faucibus dolor in elementum ullamcorper tellus ante mattis risus ac imperdiet eros eros quis risus Praesent luctus libero a diam pharetra eget placerat risus pulvinar Donec sollicitudin pulvinar velit vel pellentesque Quisque sagittis leo ac mauris congue adipiscing In tempus facilisis facilisis Aliquam erat volutpat Suspendisse sagittis libero ipsumAliquam at cursus ipsum Vivamus purus mi pretium at molestie id dictum in quam Proin egestas auctor iaculis Maecenas sodales facilisis tellus eu bibendum Vestibulum varius vehicula scelerisque Praesent condimentum varius commodo Class aptent taciti sociosqu ad litora torquent per conubia nostra per inceptos himenaeos Donec sem nisl sagittis eu euismod non tempor nec magna Fusce sed auctor nisl Phasellus porttitor sagittis est sit amet eleifend elit dignissim et Nam consectetur elementum elit non egestas Lorem ipsum dolor sit amet consectetur adipiscing elit Vestibulum a ultricies neque Integer hendrerit nisi id dolor porta quis venenatis lacus dignissim In vitae fringilla magnaFusce ultrices scelerisque felis id semper quam posuere a Sed nec erat eget velit euismod condimentum a in enim Maecenas bibendum aliquam tincidunt Mauris vestibulum neque at nulla sagittis id lacinia enim fermentum Quisque adipiscing risus nec massa auctor condimentum Mauris venenatis lacus justo eu varius odio Fusce commodo luctus felis vitae lobortis lectus facilisis id Nunc faucibus vestibulum urna et lacinia Cras ornare quam neque non gravida sapien Cras porta diam sit amet laoreet rutrum massa erat commodo diam eu rhoncus nisl massa ac metus In sem mauris venenatis nec euismod ac suscipit condimentum neque Quisque pretium blandit lectus ut aliquet neque rhoncus eu Vivamus ultrices porttitor tincidunt Curabitur ut ipsum non ipsum ultrices tincidunt Integer scelerisque augue nec nisl varius tristique Morbi condimentum rutrum sodales Pellentesque odio mauris porttitor ac sollicitudin in ultrices ut diamSed congue adipiscing orci a pellentesque Etiam quis neque eu nulla viverra egestas Ut ultricies dui non enim rhoncus laoreet Nulla molestie nibh non erat venenatis gravida Pellentesque faucibus sem sit amet risus tincidunt non ultrices diam auctor Praesent quis libero et tellus tempor molestie Mauris ullamcorper feugiat libero sed elementum Donec eget nunc eget diam hendrerit pulvinar Ut ut imperdiet enim Vestibulum sed quam lorem Nunc ipsum massa venenatis eget condimentum at ornare id ante Vestibulum ornare volutpat tincidunt Etiam a eros erat Curabitur lobortis nisi a malesuada tincidunt nisi enim congue eros in dictum elit odio at nunc Nam hendrerit porta velit a viverraEtiam vel velit urna Donec commodo aliquet magna rhoncus pretium Donec fermentum orci in diam dictum non pulvinar mi tristique Morbi urna libero sagittis vel facilisis nec ornare vitae nunc Pellentesque laoreet mi a mi condimentum sagittis Donec eleifend nisi sit amet tincidunt sollicitudin leo magna accumsan elit at adipiscing velit lacus id purus Aenean nunc sapien egestas vitae pretium viverra bibendum vel tellus Maecenas mattis dui ac justo facilisis sollicitudin Proin in mi ac lacus hendrerit congue ac vitae elit Aliquam erat volutpat In hac habitasse platea dictumst Phasellus dapibus diam vel velit consectetur tempor Maecenas viverra suscipit bibendum Sed non enim nequeCum sociis natoque penatibus et magnis dis parturient montes nascetur ridiculus mus Phasellus at odio et odio volutpat egestas Fusce non pellentesque felis Nunc fermentum posuere sem quis egestas Integer nec orci vel eros fringilla bibendum Praesent placerat molestie elit at mattis Nunc rutrum faucibus arcu non bibendum Vestibulum at sapien sit amet sem iaculis congue Morbi tempus libero vitae interdum suscipit lacus ipsum suscipit quam non pretium nulla orci eget dui Praesent et nisl turpis ultricies convallis quam In tempor urna et eros aliquet accumsan Phasellus lobortis bibendum libero sit amet viverra Aenean consectetur neque eu cursus posuere est leo molestie dui sit amet vulputate mi erat eu tortor Suspendisse arcu velit porta sit amet adipiscing sed ultrices id urna In hendrerit iaculis massa in pretium Vivamus eros augue venenatis non hendrerit a bibendum in tortor Fusce et mauris lorem vitae semper ligula Nam iaculis eros eu varius varius orci sapien rhoncus arcu et luctus urna lectus non quam Donec gravida convallis justo at bibendum Quisque non est velit sed laoreet augue - -
- - - -kind: ConfigMap -metadata: - annotations: {} - name: static-content - namespace: nginx-static-content ---- -apiVersion: apps/v1 -kind: Deployment -metadata: - annotations: {} - labels: - app: nginx-static-content - name: nginx-static-content - namespace: nginx-static-content -spec: - progressDeadlineSeconds: 600 - replicas: 1 - revisionHistoryLimit: 10 - selector: - matchLabels: - app: nginx-static-content - strategy: - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - type: RollingUpdate - template: - metadata: - labels: - app: nginx-static-content - spec: - containers: - - image: nginx:latest - imagePullPolicy: Always - name: nginx - ports: - - containerPort: 80 - protocol: TCP - resources: - requests: - cpu: 100m - memory: 100Mi - terminationMessagePath: /dev/termination-log - terminationMessagePolicy: File - volumeMounts: - - mountPath: /usr/share/nginx/html - name: static-content - - mountPath: /etc/nginx/nginx.conf - name: nginx-config - subPath: nginx.conf - dnsPolicy: ClusterFirst - nodeSelector: - cloud.google.com/compute-class: central-devops - restartPolicy: Always - schedulerName: default-scheduler - securityContext: {} - terminationGracePeriodSeconds: 30 - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: central-devops - volumes: - - configMap: - defaultMode: 420 - name: static-content - name: static-content - - configMap: - defaultMode: 420 - name: nginx-config - name: nginx-config ---- -apiVersion: v1 -kind: Service -metadata: - annotations: {} - labels: - app: nginx-static-content - name: nginx-static-content - namespace: nginx-static-content -spec: - ports: - - port: 80 - protocol: TCP - targetPort: 80 - selector: - app: nginx-static-content - sessionAffinity: None - type: ClusterIP ---- -apiVersion: projectcontour.io/v1 -kind: HTTPProxy -metadata: - annotations: - projectcontour.io/ingress.class: contour-external - labels: - app: nginx-static-content - bu: central - name: nginx-static-content - namespace: nginx-static-content -spec: - ingressClassName: contour-external - routes: - - services: - - name: nginx-static-content - port: 80 ---- diff --git a/manifests/nginx-static-content/demand/nginx-static-content-manifest.yaml b/manifests/nginx-static-content/demand/nginx-static-content-manifest.yaml deleted file mode 100644 index 2de8022..0000000 --- a/manifests/nginx-static-content/demand/nginx-static-content-manifest.yaml +++ /dev/null @@ -1,213 +0,0 @@ -# Resources from namespace "nginx-static-content" — source cluster context: gke-demand-prd-ase1a -# Recreate in a new cluster with: kubectl apply -f nginx-static-content-manifest.yaml -# NOTE: requires the Contour CRD (projectcontour.io/HTTPProxy) installed in the target cluster. ---- -apiVersion: v1 -kind: Namespace -metadata: - name: nginx-static-content ---- -apiVersion: v1 -data: - nginx.conf: | - events { - worker_connections 512; - } - http { - server { - listen 80; - - location /akamai/sureroute-test-object { - alias /usr/share/nginx/html/; - try_files $uri $uri/ /index.html; - autoindex on; - } - - location / { - return 404; - } - } - } -kind: ConfigMap -metadata: - annotations: {} - name: nginx-config - namespace: nginx-static-content ---- -apiVersion: v1 -data: - index.html: | - - - - - - Neque porro quisquam est qui dolorem ipsum quia dolor sit amet consectetur adipisci velit - - - -

Lorem Ipsum

- -

Ipsum Lorem

- -
- - Lorem ipsum dolor sit amet consectetur adipiscing elit Nullam massa enim tincidunt non hendrerit eget malesuada et nisi In hac habitasse platea dictumst Praesent nec laoreet ante Aenean tempus nisi in erat tempus tempus Vestibulum imperdiet lobortis sapien eu tempus Vivamus volutpat quam sed eros molestie vitae dignissim nulla ultricies Vivamus dictum elit velit Pellentesque pellentesque ornare ornare Mauris vel gravida sapien Praesent eleifend tristique ipsum nec tempor Vestibulum cursus eleifend tellus a egestas lectus euismod sedDuis nec massa quam Nulla porta enim ut consequat tincidunt quam tortor consequat enim eu interdum eros lorem eu turpis Cras vestibulum orci quis felis tristique quis semper sem imperdiet Sed mattis tincidunt risus scelerisque scelerisque Aliquam nisl quam bibendum quis luctus eu sodales ut felis Integer id turpis nisi Phasellus mattis nulla eu odio faucibus a auctor orci tristique Nulla ullamcorper risus nec semper accumsan libero lacus aliquet elit quis lacinia metus nunc vestibulum turpis Suspendisse vel sapien vel magna auctor aliquam Aenean fringilla fringilla metus non imperdiet Aliquam nisl lacus tempus vitae commodo non accumsan ut lectus Nam in urna eu neque pretium aliquam Maecenas sit amet urna lectus Donec vitae metus enimSed lacus nulla faucibus eget ullamcorper ut mollis at metus Vivamus tortor felis tincidunt at tristique ut tincidunt feugiat velit Ut euismod felis non urna luctus luctus Integer nec urna massa Mauris vestibulum hendrerit auctor Morbi at tellus nec arcu scelerisque rhoncus Phasellus facilisis interdum lorem vulputate posuere Nullam quis felis est Aenean metus augue tempus non ultricies et dapibus vel felis Pellentesque at augue velit Nulla erat nisi posuere eu pellentesque id pretium ac libero Phasellus tincidunt sollicitudin sapien at mollis Nullam et libero velit nec tincidunt eros Aliquam et sem elit Quisque suscipit orci enim vel aliquam nisi Suspendisse in enim a ligula blandit volutpat in id velitNam tempor neque nec ligula sollicitudin rhoncus Etiam et lorem vel odio pharetra interdum Cum sociis natoque penatibus et magnis dis parturient montes nascetur ridiculus mus In imperdiet nisi sed diam rutrum gravida in vel massa Nam ullamcorper ultrices diam vitae consequat lacus consequat consequat Curabitur laoreet leo sed tortor fringilla nec euismod libero lobortis Donec non enim lectus Suspendisse potenti In hac habitasse platea dictumst Fusce semper auctor neque nec lobortis Praesent vitae mauris turpis Lorem ipsum dolor sit amet consectetur adipiscing elit Proin sed pharetra odio Suspendisse potenti Cum sociis natoque penatibus et magnis dis parturient montes nascetur ridiculus mus Duis eget odio purus quis dapibus massaCurabitur ut dapibus eros Donec tempor felis ac facilisis bibendum nisi purus pellentesque sem sollicitudin tempor lectus nulla at mi Maecenas quis urna ut ante pulvinar pellentesque Duis auctor imperdiet suscipit Pellentesque dui nulla volutpat quis posuere a gravida ornare augue Proin nec felis pharetra magna pellentesque facilisis Curabitur lacus libero malesuada sed tincidunt ac aliquet ut tortor Etiam gravida lorem nulla consectetur eleifend risus Donec facilisis turpis laoreet imperdiet laoreet purus justo egestas nulla et hendrerit leo eros at orci Nunc vulputate mauris sit amet sapien accumsan nec euismod orci volutpat Sed ultricies velit ut lorem venenatis in convallis tellus imperdiet Aenean auctor ultrices est ultricies rhoncus Phasellus non magna a leo luctus fermentum nec fermentum eratSed faucibus nisl quis diam mollis quis varius tortor tincidunt Phasellus in turpis in tellus consectetur mollis Donec a neque id metus condimentum dignissim In hac habitasse platea dictumst Pellentesque sem nisi pulvinar nec sagittis vitae lacinia non tellus Aliquam dignissim dignissim volutpat Pellentesque ut quam et mi tincidunt varius id vel quam Duis consectetur elit ac ligula fringilla elementum In elementum tellus viverra mi vehicula vitae tempus lectus laoreet Nullam diam nibh tincidunt vitae imperdiet a luctus a felis In posuere pulvinar volutpat Pellentesque eget viverra justoNullam nec sapien at felis molestie auctor Sed dignissim erat eu nulla ullamcorper mattis Curabitur felis sem feugiat non semper ut sollicitudin sed ipsum Quisque cursus laoreet turpis sit amet molestie neque consequat at Vestibulum eu ligula quis nisl pulvinar rhoncus Praesent faucibus dolor in elementum ullamcorper tellus ante mattis risus ac imperdiet eros eros quis risus Praesent luctus libero a diam pharetra eget placerat risus pulvinar Donec sollicitudin pulvinar velit vel pellentesque Quisque sagittis leo ac mauris congue adipiscing In tempus facilisis facilisis Aliquam erat volutpat Suspendisse sagittis libero ipsumAliquam at cursus ipsum Vivamus purus mi pretium at molestie id dictum in quam Proin egestas auctor iaculis Maecenas sodales facilisis tellus eu bibendum Vestibulum varius vehicula scelerisque Praesent condimentum varius commodo Class aptent taciti sociosqu ad litora torquent per conubia nostra per inceptos himenaeos Donec sem nisl sagittis eu euismod non tempor nec magna Fusce sed auctor nisl Phasellus porttitor sagittis est sit amet eleifend elit dignissim et Nam consectetur elementum elit non egestas Lorem ipsum dolor sit amet consectetur adipiscing elit Vestibulum a ultricies neque Integer hendrerit nisi id dolor porta quis venenatis lacus dignissim In vitae fringilla magnaFusce ultrices scelerisque felis id semper quam posuere a Sed nec erat eget velit euismod condimentum a in enim Maecenas bibendum aliquam tincidunt Mauris vestibulum neque at nulla sagittis id lacinia enim fermentum Quisque adipiscing risus nec massa auctor condimentum Mauris venenatis lacus justo eu varius odio Fusce commodo luctus felis vitae lobortis lectus facilisis id Nunc faucibus vestibulum urna et lacinia Cras ornare quam neque non gravida sapien Cras porta diam sit amet laoreet rutrum massa erat commodo diam eu rhoncus nisl massa ac metus In sem mauris venenatis nec euismod ac suscipit condimentum neque Quisque pretium blandit lectus ut aliquet neque rhoncus eu Vivamus ultrices porttitor tincidunt Curabitur ut ipsum non ipsum ultrices tincidunt Integer scelerisque augue nec nisl varius tristique Morbi condimentum rutrum sodales Pellentesque odio mauris porttitor ac sollicitudin in ultrices ut diamSed congue adipiscing orci a pellentesque Etiam quis neque eu nulla viverra egestas Ut ultricies dui non enim rhoncus laoreet Nulla molestie nibh non erat venenatis gravida Pellentesque faucibus sem sit amet risus tincidunt non ultrices diam auctor Praesent quis libero et tellus tempor molestie Mauris ullamcorper feugiat libero sed elementum Donec eget nunc eget diam hendrerit pulvinar Ut ut imperdiet enim Vestibulum sed quam lorem Nunc ipsum massa venenatis eget condimentum at ornare id ante Vestibulum ornare volutpat tincidunt Etiam a eros erat Curabitur lobortis nisi a malesuada tincidunt nisi enim congue eros in dictum elit odio at nunc Nam hendrerit porta velit a viverraEtiam vel velit urna Donec commodo aliquet magna rhoncus pretium Donec fermentum orci in diam dictum non pulvinar mi tristique Morbi urna libero sagittis vel facilisis nec ornare vitae nunc Pellentesque laoreet mi a mi condimentum sagittis Donec eleifend nisi sit amet tincidunt sollicitudin leo magna accumsan elit at adipiscing velit lacus id purus Aenean nunc sapien egestas vitae pretium viverra bibendum vel tellus Maecenas mattis dui ac justo facilisis sollicitudin Proin in mi ac lacus hendrerit congue ac vitae elit Aliquam erat volutpat In hac habitasse platea dictumst Phasellus dapibus diam vel velit consectetur tempor Maecenas viverra suscipit bibendum Sed non enim nequeCum sociis natoque penatibus et magnis dis parturient montes nascetur ridiculus mus Phasellus at odio et odio volutpat egestas Fusce non pellentesque felis Nunc fermentum posuere sem quis egestas Integer nec orci vel eros fringilla bibendum Praesent placerat molestie elit at mattis Nunc rutrum faucibus arcu non bibendum Vestibulum at sapien sit amet sem iaculis congue Morbi tempus libero vitae interdum suscipit lacus ipsum suscipit quam non pretium nulla orci eget dui Praesent et nisl turpis ultricies convallis quam In tempor urna et eros aliquet accumsan Phasellus lobortis bibendum libero sit amet viverra Aenean consectetur neque eu cursus posuere est leo molestie dui sit amet vulputate mi erat eu tortor Suspendisse arcu velit porta sit amet adipiscing sed ultrices id urna In hendrerit iaculis massa in pretium Vivamus eros augue venenatis non hendrerit a bibendum in tortor Fusce et mauris lorem vitae semper ligula Nam iaculis eros eu varius varius orci sapien rhoncus arcu et luctus urna lectus non quam Donec gravida convallis justo at bibendum Quisque non est velit sed laoreet augue - -
- - - -kind: ConfigMap -metadata: - annotations: {} - name: static-content - namespace: nginx-static-content ---- -apiVersion: apps/v1 -kind: Deployment -metadata: - annotations: {} - labels: - app: nginx-static-content - name: nginx-static-content - namespace: nginx-static-content -spec: - progressDeadlineSeconds: 600 - replicas: 1 - revisionHistoryLimit: 10 - selector: - matchLabels: - app: nginx-static-content - strategy: - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - type: RollingUpdate - template: - metadata: - labels: - app: nginx-static-content - spec: - containers: - - image: nginx:latest - imagePullPolicy: Always - name: nginx - ports: - - containerPort: 80 - protocol: TCP - resources: - requests: - cpu: 100m - memory: 100Mi - terminationMessagePath: /dev/termination-log - terminationMessagePolicy: File - volumeMounts: - - mountPath: /usr/share/nginx/html - name: static-content - - mountPath: /etc/nginx/nginx.conf - name: nginx-config - subPath: nginx.conf - dnsPolicy: ClusterFirst - nodeSelector: - cloud.google.com/compute-class: demand-devops - restartPolicy: Always - schedulerName: default-scheduler - securityContext: {} - terminationGracePeriodSeconds: 30 - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: demand-devops - volumes: - - configMap: - defaultMode: 420 - name: static-content - name: static-content - - configMap: - defaultMode: 420 - name: nginx-config - name: nginx-config ---- -apiVersion: v1 -kind: Service -metadata: - annotations: {} - labels: - app: nginx-static-content - name: nginx-static-content - namespace: nginx-static-content -spec: - ports: - - port: 80 - protocol: TCP - targetPort: 80 - selector: - app: nginx-static-content - sessionAffinity: None - type: ClusterIP ---- -apiVersion: projectcontour.io/v1 -kind: HTTPProxy -metadata: - annotations: - projectcontour.io/ingress.class: contour-external - labels: - app: nginx-static-content - bu: demand - name: nginx-static-content - namespace: nginx-static-content -spec: - ingressClassName: contour-external - routes: - - services: - - name: nginx-static-content - port: 80 ---- diff --git a/manifests/nginx-static-content/farmiso/nginx-static-content-manifest.yaml b/manifests/nginx-static-content/farmiso/nginx-static-content-manifest.yaml deleted file mode 100644 index d3a58a1..0000000 --- a/manifests/nginx-static-content/farmiso/nginx-static-content-manifest.yaml +++ /dev/null @@ -1,213 +0,0 @@ -# Resources from namespace "nginx-static-content" — source cluster context: kc-farmiso-prd-ase1 -# Recreate in a new cluster with: kubectl apply -f nginx-static-content-manifest.yaml -# NOTE: requires the Contour CRD (projectcontour.io/HTTPProxy) installed in the target cluster. ---- -apiVersion: v1 -kind: Namespace -metadata: - name: nginx-static-content ---- -apiVersion: v1 -data: - nginx.conf: | - events { - worker_connections 512; - } - http { - server { - listen 80; - - location /akamai/sureroute-test-object { - alias /usr/share/nginx/html/; - try_files $uri $uri/ /index.html; - autoindex on; - } - - location / { - return 404; - } - } - } -kind: ConfigMap -metadata: - annotations: {} - name: nginx-config - namespace: nginx-static-content ---- -apiVersion: v1 -data: - index.html: | - - - - - - Neque porro quisquam est qui dolorem ipsum quia dolor sit amet consectetur adipisci velit - - - -

Lorem Ipsum

- -

Ipsum Lorem

- -
- - Lorem ipsum dolor sit amet consectetur adipiscing elit Nullam massa enim tincidunt non hendrerit eget malesuada et nisi In hac habitasse platea dictumst Praesent nec laoreet ante Aenean tempus nisi in erat tempus tempus Vestibulum imperdiet lobortis sapien eu tempus Vivamus volutpat quam sed eros molestie vitae dignissim nulla ultricies Vivamus dictum elit velit Pellentesque pellentesque ornare ornare Mauris vel gravida sapien Praesent eleifend tristique ipsum nec tempor Vestibulum cursus eleifend tellus a egestas lectus euismod sedDuis nec massa quam Nulla porta enim ut consequat tincidunt quam tortor consequat enim eu interdum eros lorem eu turpis Cras vestibulum orci quis felis tristique quis semper sem imperdiet Sed mattis tincidunt risus scelerisque scelerisque Aliquam nisl quam bibendum quis luctus eu sodales ut felis Integer id turpis nisi Phasellus mattis nulla eu odio faucibus a auctor orci tristique Nulla ullamcorper risus nec semper accumsan libero lacus aliquet elit quis lacinia metus nunc vestibulum turpis Suspendisse vel sapien vel magna auctor aliquam Aenean fringilla fringilla metus non imperdiet Aliquam nisl lacus tempus vitae commodo non accumsan ut lectus Nam in urna eu neque pretium aliquam Maecenas sit amet urna lectus Donec vitae metus enimSed lacus nulla faucibus eget ullamcorper ut mollis at metus Vivamus tortor felis tincidunt at tristique ut tincidunt feugiat velit Ut euismod felis non urna luctus luctus Integer nec urna massa Mauris vestibulum hendrerit auctor Morbi at tellus nec arcu scelerisque rhoncus Phasellus facilisis interdum lorem vulputate posuere Nullam quis felis est Aenean metus augue tempus non ultricies et dapibus vel felis Pellentesque at augue velit Nulla erat nisi posuere eu pellentesque id pretium ac libero Phasellus tincidunt sollicitudin sapien at mollis Nullam et libero velit nec tincidunt eros Aliquam et sem elit Quisque suscipit orci enim vel aliquam nisi Suspendisse in enim a ligula blandit volutpat in id velitNam tempor neque nec ligula sollicitudin rhoncus Etiam et lorem vel odio pharetra interdum Cum sociis natoque penatibus et magnis dis parturient montes nascetur ridiculus mus In imperdiet nisi sed diam rutrum gravida in vel massa Nam ullamcorper ultrices diam vitae consequat lacus consequat consequat Curabitur laoreet leo sed tortor fringilla nec euismod libero lobortis Donec non enim lectus Suspendisse potenti In hac habitasse platea dictumst Fusce semper auctor neque nec lobortis Praesent vitae mauris turpis Lorem ipsum dolor sit amet consectetur adipiscing elit Proin sed pharetra odio Suspendisse potenti Cum sociis natoque penatibus et magnis dis parturient montes nascetur ridiculus mus Duis eget odio purus quis dapibus massaCurabitur ut dapibus eros Donec tempor felis ac facilisis bibendum nisi purus pellentesque sem sollicitudin tempor lectus nulla at mi Maecenas quis urna ut ante pulvinar pellentesque Duis auctor imperdiet suscipit Pellentesque dui nulla volutpat quis posuere a gravida ornare augue Proin nec felis pharetra magna pellentesque facilisis Curabitur lacus libero malesuada sed tincidunt ac aliquet ut tortor Etiam gravida lorem nulla consectetur eleifend risus Donec facilisis turpis laoreet imperdiet laoreet purus justo egestas nulla et hendrerit leo eros at orci Nunc vulputate mauris sit amet sapien accumsan nec euismod orci volutpat Sed ultricies velit ut lorem venenatis in convallis tellus imperdiet Aenean auctor ultrices est ultricies rhoncus Phasellus non magna a leo luctus fermentum nec fermentum eratSed faucibus nisl quis diam mollis quis varius tortor tincidunt Phasellus in turpis in tellus consectetur mollis Donec a neque id metus condimentum dignissim In hac habitasse platea dictumst Pellentesque sem nisi pulvinar nec sagittis vitae lacinia non tellus Aliquam dignissim dignissim volutpat Pellentesque ut quam et mi tincidunt varius id vel quam Duis consectetur elit ac ligula fringilla elementum In elementum tellus viverra mi vehicula vitae tempus lectus laoreet Nullam diam nibh tincidunt vitae imperdiet a luctus a felis In posuere pulvinar volutpat Pellentesque eget viverra justoNullam nec sapien at felis molestie auctor Sed dignissim erat eu nulla ullamcorper mattis Curabitur felis sem feugiat non semper ut sollicitudin sed ipsum Quisque cursus laoreet turpis sit amet molestie neque consequat at Vestibulum eu ligula quis nisl pulvinar rhoncus Praesent faucibus dolor in elementum ullamcorper tellus ante mattis risus ac imperdiet eros eros quis risus Praesent luctus libero a diam pharetra eget placerat risus pulvinar Donec sollicitudin pulvinar velit vel pellentesque Quisque sagittis leo ac mauris congue adipiscing In tempus facilisis facilisis Aliquam erat volutpat Suspendisse sagittis libero ipsumAliquam at cursus ipsum Vivamus purus mi pretium at molestie id dictum in quam Proin egestas auctor iaculis Maecenas sodales facilisis tellus eu bibendum Vestibulum varius vehicula scelerisque Praesent condimentum varius commodo Class aptent taciti sociosqu ad litora torquent per conubia nostra per inceptos himenaeos Donec sem nisl sagittis eu euismod non tempor nec magna Fusce sed auctor nisl Phasellus porttitor sagittis est sit amet eleifend elit dignissim et Nam consectetur elementum elit non egestas Lorem ipsum dolor sit amet consectetur adipiscing elit Vestibulum a ultricies neque Integer hendrerit nisi id dolor porta quis venenatis lacus dignissim In vitae fringilla magnaFusce ultrices scelerisque felis id semper quam posuere a Sed nec erat eget velit euismod condimentum a in enim Maecenas bibendum aliquam tincidunt Mauris vestibulum neque at nulla sagittis id lacinia enim fermentum Quisque adipiscing risus nec massa auctor condimentum Mauris venenatis lacus justo eu varius odio Fusce commodo luctus felis vitae lobortis lectus facilisis id Nunc faucibus vestibulum urna et lacinia Cras ornare quam neque non gravida sapien Cras porta diam sit amet laoreet rutrum massa erat commodo diam eu rhoncus nisl massa ac metus In sem mauris venenatis nec euismod ac suscipit condimentum neque Quisque pretium blandit lectus ut aliquet neque rhoncus eu Vivamus ultrices porttitor tincidunt Curabitur ut ipsum non ipsum ultrices tincidunt Integer scelerisque augue nec nisl varius tristique Morbi condimentum rutrum sodales Pellentesque odio mauris porttitor ac sollicitudin in ultrices ut diamSed congue adipiscing orci a pellentesque Etiam quis neque eu nulla viverra egestas Ut ultricies dui non enim rhoncus laoreet Nulla molestie nibh non erat venenatis gravida Pellentesque faucibus sem sit amet risus tincidunt non ultrices diam auctor Praesent quis libero et tellus tempor molestie Mauris ullamcorper feugiat libero sed elementum Donec eget nunc eget diam hendrerit pulvinar Ut ut imperdiet enim Vestibulum sed quam lorem Nunc ipsum massa venenatis eget condimentum at ornare id ante Vestibulum ornare volutpat tincidunt Etiam a eros erat Curabitur lobortis nisi a malesuada tincidunt nisi enim congue eros in dictum elit odio at nunc Nam hendrerit porta velit a viverraEtiam vel velit urna Donec commodo aliquet magna rhoncus pretium Donec fermentum orci in diam dictum non pulvinar mi tristique Morbi urna libero sagittis vel facilisis nec ornare vitae nunc Pellentesque laoreet mi a mi condimentum sagittis Donec eleifend nisi sit amet tincidunt sollicitudin leo magna accumsan elit at adipiscing velit lacus id purus Aenean nunc sapien egestas vitae pretium viverra bibendum vel tellus Maecenas mattis dui ac justo facilisis sollicitudin Proin in mi ac lacus hendrerit congue ac vitae elit Aliquam erat volutpat In hac habitasse platea dictumst Phasellus dapibus diam vel velit consectetur tempor Maecenas viverra suscipit bibendum Sed non enim nequeCum sociis natoque penatibus et magnis dis parturient montes nascetur ridiculus mus Phasellus at odio et odio volutpat egestas Fusce non pellentesque felis Nunc fermentum posuere sem quis egestas Integer nec orci vel eros fringilla bibendum Praesent placerat molestie elit at mattis Nunc rutrum faucibus arcu non bibendum Vestibulum at sapien sit amet sem iaculis congue Morbi tempus libero vitae interdum suscipit lacus ipsum suscipit quam non pretium nulla orci eget dui Praesent et nisl turpis ultricies convallis quam In tempor urna et eros aliquet accumsan Phasellus lobortis bibendum libero sit amet viverra Aenean consectetur neque eu cursus posuere est leo molestie dui sit amet vulputate mi erat eu tortor Suspendisse arcu velit porta sit amet adipiscing sed ultrices id urna In hendrerit iaculis massa in pretium Vivamus eros augue venenatis non hendrerit a bibendum in tortor Fusce et mauris lorem vitae semper ligula Nam iaculis eros eu varius varius orci sapien rhoncus arcu et luctus urna lectus non quam Donec gravida convallis justo at bibendum Quisque non est velit sed laoreet augue - -
- - - -kind: ConfigMap -metadata: - annotations: {} - name: static-content - namespace: nginx-static-content ---- -apiVersion: apps/v1 -kind: Deployment -metadata: - annotations: {} - labels: - app: nginx-static-content - name: nginx-static-content - namespace: nginx-static-content -spec: - progressDeadlineSeconds: 600 - replicas: 1 - revisionHistoryLimit: 10 - selector: - matchLabels: - app: nginx-static-content - strategy: - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - type: RollingUpdate - template: - metadata: - labels: - app: nginx-static-content - spec: - containers: - - image: nginx:latest - imagePullPolicy: Always - name: nginx - ports: - - containerPort: 80 - protocol: TCP - resources: - requests: - cpu: 100m - memory: 100Mi - terminationMessagePath: /dev/termination-log - terminationMessagePolicy: File - volumeMounts: - - mountPath: /usr/share/nginx/html - name: static-content - - mountPath: /etc/nginx/nginx.conf - name: nginx-config - subPath: nginx.conf - dnsPolicy: ClusterFirst - nodeSelector: - cloud.google.com/compute-class: farmiso-devops - restartPolicy: Always - schedulerName: default-scheduler - securityContext: {} - terminationGracePeriodSeconds: 30 - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: farmiso-devops - volumes: - - configMap: - defaultMode: 420 - name: static-content - name: static-content - - configMap: - defaultMode: 420 - name: nginx-config - name: nginx-config ---- -apiVersion: v1 -kind: Service -metadata: - annotations: {} - labels: - app: nginx-static-content - name: nginx-static-content - namespace: nginx-static-content -spec: - ports: - - port: 80 - protocol: TCP - targetPort: 80 - selector: - app: nginx-static-content - sessionAffinity: None - type: ClusterIP ---- -apiVersion: projectcontour.io/v1 -kind: HTTPProxy -metadata: - annotations: - projectcontour.io/ingress.class: contour-external - labels: - app: nginx-static-content - bu: farmiso - name: nginx-static-content - namespace: nginx-static-content -spec: - ingressClassName: contour-external - routes: - - services: - - name: nginx-static-content - port: 80 ---- diff --git a/manifests/nginx-static-content/supply/nginx-static-content-manifest.yaml b/manifests/nginx-static-content/supply/nginx-static-content-manifest.yaml deleted file mode 100644 index 6bc8a05..0000000 --- a/manifests/nginx-static-content/supply/nginx-static-content-manifest.yaml +++ /dev/null @@ -1,213 +0,0 @@ -# Resources from namespace "nginx-static-content" — source cluster context: gke-supply-prd-ase1a -# Recreate in a new cluster with: kubectl apply -f nginx-static-content-manifest.yaml -# NOTE: requires the Contour CRD (projectcontour.io/HTTPProxy) installed in the target cluster. ---- -apiVersion: v1 -kind: Namespace -metadata: - name: nginx-static-content ---- -apiVersion: v1 -data: - nginx.conf: | - events { - worker_connections 512; - } - http { - server { - listen 80; - - location /akamai/sureroute-test-object { - alias /usr/share/nginx/html/; - try_files $uri $uri/ /index.html; - autoindex on; - } - - location / { - return 404; - } - } - } -kind: ConfigMap -metadata: - annotations: {} - name: nginx-config - namespace: nginx-static-content ---- -apiVersion: v1 -data: - index.html: | - - - - - - Neque porro quisquam est qui dolorem ipsum quia dolor sit amet consectetur adipisci velit - - - -

Lorem Ipsum

- -

Ipsum Lorem

- -
- - Lorem ipsum dolor sit amet consectetur adipiscing elit Nullam massa enim tincidunt non hendrerit eget malesuada et nisi In hac habitasse platea dictumst Praesent nec laoreet ante Aenean tempus nisi in erat tempus tempus Vestibulum imperdiet lobortis sapien eu tempus Vivamus volutpat quam sed eros molestie vitae dignissim nulla ultricies Vivamus dictum elit velit Pellentesque pellentesque ornare ornare Mauris vel gravida sapien Praesent eleifend tristique ipsum nec tempor Vestibulum cursus eleifend tellus a egestas lectus euismod sedDuis nec massa quam Nulla porta enim ut consequat tincidunt quam tortor consequat enim eu interdum eros lorem eu turpis Cras vestibulum orci quis felis tristique quis semper sem imperdiet Sed mattis tincidunt risus scelerisque scelerisque Aliquam nisl quam bibendum quis luctus eu sodales ut felis Integer id turpis nisi Phasellus mattis nulla eu odio faucibus a auctor orci tristique Nulla ullamcorper risus nec semper accumsan libero lacus aliquet elit quis lacinia metus nunc vestibulum turpis Suspendisse vel sapien vel magna auctor aliquam Aenean fringilla fringilla metus non imperdiet Aliquam nisl lacus tempus vitae commodo non accumsan ut lectus Nam in urna eu neque pretium aliquam Maecenas sit amet urna lectus Donec vitae metus enimSed lacus nulla faucibus eget ullamcorper ut mollis at metus Vivamus tortor felis tincidunt at tristique ut tincidunt feugiat velit Ut euismod felis non urna luctus luctus Integer nec urna massa Mauris vestibulum hendrerit auctor Morbi at tellus nec arcu scelerisque rhoncus Phasellus facilisis interdum lorem vulputate posuere Nullam quis felis est Aenean metus augue tempus non ultricies et dapibus vel felis Pellentesque at augue velit Nulla erat nisi posuere eu pellentesque id pretium ac libero Phasellus tincidunt sollicitudin sapien at mollis Nullam et libero velit nec tincidunt eros Aliquam et sem elit Quisque suscipit orci enim vel aliquam nisi Suspendisse in enim a ligula blandit volutpat in id velitNam tempor neque nec ligula sollicitudin rhoncus Etiam et lorem vel odio pharetra interdum Cum sociis natoque penatibus et magnis dis parturient montes nascetur ridiculus mus In imperdiet nisi sed diam rutrum gravida in vel massa Nam ullamcorper ultrices diam vitae consequat lacus consequat consequat Curabitur laoreet leo sed tortor fringilla nec euismod libero lobortis Donec non enim lectus Suspendisse potenti In hac habitasse platea dictumst Fusce semper auctor neque nec lobortis Praesent vitae mauris turpis Lorem ipsum dolor sit amet consectetur adipiscing elit Proin sed pharetra odio Suspendisse potenti Cum sociis natoque penatibus et magnis dis parturient montes nascetur ridiculus mus Duis eget odio purus quis dapibus massaCurabitur ut dapibus eros Donec tempor felis ac facilisis bibendum nisi purus pellentesque sem sollicitudin tempor lectus nulla at mi Maecenas quis urna ut ante pulvinar pellentesque Duis auctor imperdiet suscipit Pellentesque dui nulla volutpat quis posuere a gravida ornare augue Proin nec felis pharetra magna pellentesque facilisis Curabitur lacus libero malesuada sed tincidunt ac aliquet ut tortor Etiam gravida lorem nulla consectetur eleifend risus Donec facilisis turpis laoreet imperdiet laoreet purus justo egestas nulla et hendrerit leo eros at orci Nunc vulputate mauris sit amet sapien accumsan nec euismod orci volutpat Sed ultricies velit ut lorem venenatis in convallis tellus imperdiet Aenean auctor ultrices est ultricies rhoncus Phasellus non magna a leo luctus fermentum nec fermentum eratSed faucibus nisl quis diam mollis quis varius tortor tincidunt Phasellus in turpis in tellus consectetur mollis Donec a neque id metus condimentum dignissim In hac habitasse platea dictumst Pellentesque sem nisi pulvinar nec sagittis vitae lacinia non tellus Aliquam dignissim dignissim volutpat Pellentesque ut quam et mi tincidunt varius id vel quam Duis consectetur elit ac ligula fringilla elementum In elementum tellus viverra mi vehicula vitae tempus lectus laoreet Nullam diam nibh tincidunt vitae imperdiet a luctus a felis In posuere pulvinar volutpat Pellentesque eget viverra justoNullam nec sapien at felis molestie auctor Sed dignissim erat eu nulla ullamcorper mattis Curabitur felis sem feugiat non semper ut sollicitudin sed ipsum Quisque cursus laoreet turpis sit amet molestie neque consequat at Vestibulum eu ligula quis nisl pulvinar rhoncus Praesent faucibus dolor in elementum ullamcorper tellus ante mattis risus ac imperdiet eros eros quis risus Praesent luctus libero a diam pharetra eget placerat risus pulvinar Donec sollicitudin pulvinar velit vel pellentesque Quisque sagittis leo ac mauris congue adipiscing In tempus facilisis facilisis Aliquam erat volutpat Suspendisse sagittis libero ipsumAliquam at cursus ipsum Vivamus purus mi pretium at molestie id dictum in quam Proin egestas auctor iaculis Maecenas sodales facilisis tellus eu bibendum Vestibulum varius vehicula scelerisque Praesent condimentum varius commodo Class aptent taciti sociosqu ad litora torquent per conubia nostra per inceptos himenaeos Donec sem nisl sagittis eu euismod non tempor nec magna Fusce sed auctor nisl Phasellus porttitor sagittis est sit amet eleifend elit dignissim et Nam consectetur elementum elit non egestas Lorem ipsum dolor sit amet consectetur adipiscing elit Vestibulum a ultricies neque Integer hendrerit nisi id dolor porta quis venenatis lacus dignissim In vitae fringilla magnaFusce ultrices scelerisque felis id semper quam posuere a Sed nec erat eget velit euismod condimentum a in enim Maecenas bibendum aliquam tincidunt Mauris vestibulum neque at nulla sagittis id lacinia enim fermentum Quisque adipiscing risus nec massa auctor condimentum Mauris venenatis lacus justo eu varius odio Fusce commodo luctus felis vitae lobortis lectus facilisis id Nunc faucibus vestibulum urna et lacinia Cras ornare quam neque non gravida sapien Cras porta diam sit amet laoreet rutrum massa erat commodo diam eu rhoncus nisl massa ac metus In sem mauris venenatis nec euismod ac suscipit condimentum neque Quisque pretium blandit lectus ut aliquet neque rhoncus eu Vivamus ultrices porttitor tincidunt Curabitur ut ipsum non ipsum ultrices tincidunt Integer scelerisque augue nec nisl varius tristique Morbi condimentum rutrum sodales Pellentesque odio mauris porttitor ac sollicitudin in ultrices ut diamSed congue adipiscing orci a pellentesque Etiam quis neque eu nulla viverra egestas Ut ultricies dui non enim rhoncus laoreet Nulla molestie nibh non erat venenatis gravida Pellentesque faucibus sem sit amet risus tincidunt non ultrices diam auctor Praesent quis libero et tellus tempor molestie Mauris ullamcorper feugiat libero sed elementum Donec eget nunc eget diam hendrerit pulvinar Ut ut imperdiet enim Vestibulum sed quam lorem Nunc ipsum massa venenatis eget condimentum at ornare id ante Vestibulum ornare volutpat tincidunt Etiam a eros erat Curabitur lobortis nisi a malesuada tincidunt nisi enim congue eros in dictum elit odio at nunc Nam hendrerit porta velit a viverraEtiam vel velit urna Donec commodo aliquet magna rhoncus pretium Donec fermentum orci in diam dictum non pulvinar mi tristique Morbi urna libero sagittis vel facilisis nec ornare vitae nunc Pellentesque laoreet mi a mi condimentum sagittis Donec eleifend nisi sit amet tincidunt sollicitudin leo magna accumsan elit at adipiscing velit lacus id purus Aenean nunc sapien egestas vitae pretium viverra bibendum vel tellus Maecenas mattis dui ac justo facilisis sollicitudin Proin in mi ac lacus hendrerit congue ac vitae elit Aliquam erat volutpat In hac habitasse platea dictumst Phasellus dapibus diam vel velit consectetur tempor Maecenas viverra suscipit bibendum Sed non enim nequeCum sociis natoque penatibus et magnis dis parturient montes nascetur ridiculus mus Phasellus at odio et odio volutpat egestas Fusce non pellentesque felis Nunc fermentum posuere sem quis egestas Integer nec orci vel eros fringilla bibendum Praesent placerat molestie elit at mattis Nunc rutrum faucibus arcu non bibendum Vestibulum at sapien sit amet sem iaculis congue Morbi tempus libero vitae interdum suscipit lacus ipsum suscipit quam non pretium nulla orci eget dui Praesent et nisl turpis ultricies convallis quam In tempor urna et eros aliquet accumsan Phasellus lobortis bibendum libero sit amet viverra Aenean consectetur neque eu cursus posuere est leo molestie dui sit amet vulputate mi erat eu tortor Suspendisse arcu velit porta sit amet adipiscing sed ultrices id urna In hendrerit iaculis massa in pretium Vivamus eros augue venenatis non hendrerit a bibendum in tortor Fusce et mauris lorem vitae semper ligula Nam iaculis eros eu varius varius orci sapien rhoncus arcu et luctus urna lectus non quam Donec gravida convallis justo at bibendum Quisque non est velit sed laoreet augue - -
- - - -kind: ConfigMap -metadata: - annotations: {} - name: static-content - namespace: nginx-static-content ---- -apiVersion: apps/v1 -kind: Deployment -metadata: - annotations: {} - labels: - app: nginx-static-content - name: nginx-static-content - namespace: nginx-static-content -spec: - progressDeadlineSeconds: 600 - replicas: 1 - revisionHistoryLimit: 10 - selector: - matchLabels: - app: nginx-static-content - strategy: - rollingUpdate: - maxSurge: 25% - maxUnavailable: 25% - type: RollingUpdate - template: - metadata: - labels: - app: nginx-static-content - spec: - containers: - - image: nginx:latest - imagePullPolicy: Always - name: nginx - ports: - - containerPort: 80 - protocol: TCP - resources: - requests: - cpu: 100m - memory: 100Mi - terminationMessagePath: /dev/termination-log - terminationMessagePolicy: File - volumeMounts: - - mountPath: /usr/share/nginx/html - name: static-content - - mountPath: /etc/nginx/nginx.conf - name: nginx-config - subPath: nginx.conf - dnsPolicy: ClusterFirst - nodeSelector: - cloud.google.com/compute-class: supply-devops - restartPolicy: Always - schedulerName: default-scheduler - securityContext: {} - terminationGracePeriodSeconds: 30 - tolerations: - - effect: NoSchedule - key: cloud.google.com/compute-class - operator: Equal - value: supply-devops - volumes: - - configMap: - defaultMode: 420 - name: static-content - name: static-content - - configMap: - defaultMode: 420 - name: nginx-config - name: nginx-config ---- -apiVersion: v1 -kind: Service -metadata: - annotations: {} - labels: - app: nginx-static-content - name: nginx-static-content - namespace: nginx-static-content -spec: - ports: - - port: 80 - protocol: TCP - targetPort: 80 - selector: - app: nginx-static-content - sessionAffinity: None - type: ClusterIP ---- -apiVersion: projectcontour.io/v1 -kind: HTTPProxy -metadata: - annotations: - projectcontour.io/ingress.class: contour-external - labels: - app: nginx-static-content - bu: supply - name: nginx-static-content - namespace: nginx-static-content -spec: - ingressClassName: contour-external - routes: - - services: - - name: nginx-static-content - port: 80 ---- diff --git a/manifests/priorityclass/gke-central-prd-ase1a/priorityclass-high.yaml b/manifests/priorityclass/gke-central-prd-ase1a/priorityclass-high.yaml deleted file mode 100644 index 038ae8e..0000000 --- a/manifests/priorityclass/gke-central-prd-ase1a/priorityclass-high.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is high priority and should be used for app service pods only. -kind: PriorityClass -metadata: - name: high-priority - namespace: kube-system -value: 1000000 diff --git a/manifests/priorityclass/gke-central-prd-ase1a/priorityclass-low.yaml b/manifests/priorityclass/gke-central-prd-ase1a/priorityclass-low.yaml deleted file mode 100644 index d3446cb..0000000 --- a/manifests/priorityclass/gke-central-prd-ase1a/priorityclass-low.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is low priority and should be used for non-app pods only. -kind: PriorityClass -metadata: - name: low-priority - namespace: kube-system -value: 1000 \ No newline at end of file diff --git a/manifests/priorityclass/gke-dataengg-prd-ase1a/priorityclass-high.yaml b/manifests/priorityclass/gke-dataengg-prd-ase1a/priorityclass-high.yaml deleted file mode 100644 index 038ae8e..0000000 --- a/manifests/priorityclass/gke-dataengg-prd-ase1a/priorityclass-high.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is high priority and should be used for app service pods only. -kind: PriorityClass -metadata: - name: high-priority - namespace: kube-system -value: 1000000 diff --git a/manifests/priorityclass/gke-dataengg-prd-ase1a/priorityclass-low.yaml b/manifests/priorityclass/gke-dataengg-prd-ase1a/priorityclass-low.yaml deleted file mode 100644 index d3446cb..0000000 --- a/manifests/priorityclass/gke-dataengg-prd-ase1a/priorityclass-low.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is low priority and should be used for non-app pods only. -kind: PriorityClass -metadata: - name: low-priority - namespace: kube-system -value: 1000 \ No newline at end of file diff --git a/manifests/priorityclass/gke-datascience-prd-ase1a/priorityclass-high.yaml b/manifests/priorityclass/gke-datascience-prd-ase1a/priorityclass-high.yaml deleted file mode 100644 index 038ae8e..0000000 --- a/manifests/priorityclass/gke-datascience-prd-ase1a/priorityclass-high.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is high priority and should be used for app service pods only. -kind: PriorityClass -metadata: - name: high-priority - namespace: kube-system -value: 1000000 diff --git a/manifests/priorityclass/gke-datascience-prd-ase1a/priorityclass-low.yaml b/manifests/priorityclass/gke-datascience-prd-ase1a/priorityclass-low.yaml deleted file mode 100644 index d3446cb..0000000 --- a/manifests/priorityclass/gke-datascience-prd-ase1a/priorityclass-low.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is low priority and should be used for non-app pods only. -kind: PriorityClass -metadata: - name: low-priority - namespace: kube-system -value: 1000 \ No newline at end of file diff --git a/manifests/priorityclass/gke-datascience-prd-ase1a/spot-termination-handler.yaml b/manifests/priorityclass/gke-datascience-prd-ase1a/spot-termination-handler.yaml deleted file mode 100644 index 03df7af..0000000 --- a/manifests/priorityclass/gke-datascience-prd-ase1a/spot-termination-handler.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -kind: PriorityClass -metadata: - name: spot-termination-handler -description: | - High priority for spot termination handler DaemonSet. Ensures handler pod is terminated after regular workloads during node shutdown, giving it time to drain the node. -preemptionPolicy: PreemptLowerPriority -value: 1000000 diff --git a/manifests/priorityclass/gke-dsgpu-prd-ase1a/priorityclass-high.yaml b/manifests/priorityclass/gke-dsgpu-prd-ase1a/priorityclass-high.yaml deleted file mode 100644 index 038ae8e..0000000 --- a/manifests/priorityclass/gke-dsgpu-prd-ase1a/priorityclass-high.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is high priority and should be used for app service pods only. -kind: PriorityClass -metadata: - name: high-priority - namespace: kube-system -value: 1000000 diff --git a/manifests/priorityclass/gke-dsgpu-prd-ase1a/priorityclass-low.yaml b/manifests/priorityclass/gke-dsgpu-prd-ase1a/priorityclass-low.yaml deleted file mode 100644 index d3446cb..0000000 --- a/manifests/priorityclass/gke-dsgpu-prd-ase1a/priorityclass-low.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is low priority and should be used for non-app pods only. -kind: PriorityClass -metadata: - name: low-priority - namespace: kube-system -value: 1000 \ No newline at end of file diff --git a/manifests/priorityclass/gke-farmiso-prd-ase1a/priorityclass-high.yaml b/manifests/priorityclass/gke-farmiso-prd-ase1a/priorityclass-high.yaml deleted file mode 100644 index 038ae8e..0000000 --- a/manifests/priorityclass/gke-farmiso-prd-ase1a/priorityclass-high.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is high priority and should be used for app service pods only. -kind: PriorityClass -metadata: - name: high-priority - namespace: kube-system -value: 1000000 diff --git a/manifests/priorityclass/k8s-central-prd-ase1/priorityclass-high.yaml b/manifests/priorityclass/k8s-central-prd-ase1/priorityclass-high.yaml deleted file mode 100644 index 038ae8e..0000000 --- a/manifests/priorityclass/k8s-central-prd-ase1/priorityclass-high.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is high priority and should be used for app service pods only. -kind: PriorityClass -metadata: - name: high-priority - namespace: kube-system -value: 1000000 diff --git a/manifests/priorityclass/k8s-central-prd-ase1/priorityclass-low.yaml b/manifests/priorityclass/k8s-central-prd-ase1/priorityclass-low.yaml deleted file mode 100644 index d3446cb..0000000 --- a/manifests/priorityclass/k8s-central-prd-ase1/priorityclass-low.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is low priority and should be used for non-app pods only. -kind: PriorityClass -metadata: - name: low-priority - namespace: kube-system -value: 1000 \ No newline at end of file diff --git a/manifests/priorityclass/k8s-dataengg-prd-ase1/priorityclass-high.yaml b/manifests/priorityclass/k8s-dataengg-prd-ase1/priorityclass-high.yaml deleted file mode 100644 index 038ae8e..0000000 --- a/manifests/priorityclass/k8s-dataengg-prd-ase1/priorityclass-high.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is high priority and should be used for app service pods only. -kind: PriorityClass -metadata: - name: high-priority - namespace: kube-system -value: 1000000 diff --git a/manifests/priorityclass/k8s-dataengg-prd-ase1/priorityclass-low.yaml b/manifests/priorityclass/k8s-dataengg-prd-ase1/priorityclass-low.yaml deleted file mode 100644 index d3446cb..0000000 --- a/manifests/priorityclass/k8s-dataengg-prd-ase1/priorityclass-low.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is low priority and should be used for non-app pods only. -kind: PriorityClass -metadata: - name: low-priority - namespace: kube-system -value: 1000 \ No newline at end of file diff --git a/manifests/priorityclass/k8s-datascience-prd-ase1/priorityclass-high.yaml b/manifests/priorityclass/k8s-datascience-prd-ase1/priorityclass-high.yaml deleted file mode 100644 index 038ae8e..0000000 --- a/manifests/priorityclass/k8s-datascience-prd-ase1/priorityclass-high.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is high priority and should be used for app service pods only. -kind: PriorityClass -metadata: - name: high-priority - namespace: kube-system -value: 1000000 diff --git a/manifests/priorityclass/k8s-datascience-prd-ase1/priorityclass-low.yaml b/manifests/priorityclass/k8s-datascience-prd-ase1/priorityclass-low.yaml deleted file mode 100644 index d3446cb..0000000 --- a/manifests/priorityclass/k8s-datascience-prd-ase1/priorityclass-low.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is low priority and should be used for non-app pods only. -kind: PriorityClass -metadata: - name: low-priority - namespace: kube-system -value: 1000 \ No newline at end of file diff --git a/manifests/priorityclass/k8s-demand-prd-ase1/priorityclass-high.yaml b/manifests/priorityclass/k8s-demand-prd-ase1/priorityclass-high.yaml deleted file mode 100644 index 038ae8e..0000000 --- a/manifests/priorityclass/k8s-demand-prd-ase1/priorityclass-high.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is high priority and should be used for app service pods only. -kind: PriorityClass -metadata: - name: high-priority - namespace: kube-system -value: 1000000 diff --git a/manifests/priorityclass/k8s-demand-prd-ase1/priorityclass-low.yaml b/manifests/priorityclass/k8s-demand-prd-ase1/priorityclass-low.yaml deleted file mode 100644 index d3446cb..0000000 --- a/manifests/priorityclass/k8s-demand-prd-ase1/priorityclass-low.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is low priority and should be used for non-app pods only. -kind: PriorityClass -metadata: - name: low-priority - namespace: kube-system -value: 1000 \ No newline at end of file diff --git a/manifests/priorityclass/k8s-farmiso-prd-ase1/priorityclass-high.yaml b/manifests/priorityclass/k8s-farmiso-prd-ase1/priorityclass-high.yaml deleted file mode 100644 index 038ae8e..0000000 --- a/manifests/priorityclass/k8s-farmiso-prd-ase1/priorityclass-high.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is high priority and should be used for app service pods only. -kind: PriorityClass -metadata: - name: high-priority - namespace: kube-system -value: 1000000 diff --git a/manifests/priorityclass/k8s-farmiso-prd-ase1/priorityclass-low.yaml b/manifests/priorityclass/k8s-farmiso-prd-ase1/priorityclass-low.yaml deleted file mode 100644 index d3446cb..0000000 --- a/manifests/priorityclass/k8s-farmiso-prd-ase1/priorityclass-low.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is low priority and should be used for non-app pods only. -kind: PriorityClass -metadata: - name: low-priority - namespace: kube-system -value: 1000 \ No newline at end of file diff --git a/manifests/priorityclass/k8s-sec-admin-ase1/priorityclass-high.yaml b/manifests/priorityclass/k8s-sec-admin-ase1/priorityclass-high.yaml deleted file mode 100644 index 038ae8e..0000000 --- a/manifests/priorityclass/k8s-sec-admin-ase1/priorityclass-high.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is high priority and should be used for app service pods only. -kind: PriorityClass -metadata: - name: high-priority - namespace: kube-system -value: 1000000 diff --git a/manifests/priorityclass/k8s-sec-admin-ase1/priorityclass-low.yaml b/manifests/priorityclass/k8s-sec-admin-ase1/priorityclass-low.yaml deleted file mode 100644 index d3446cb..0000000 --- a/manifests/priorityclass/k8s-sec-admin-ase1/priorityclass-low.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is low priority and should be used for non-app pods only. -kind: PriorityClass -metadata: - name: low-priority - namespace: kube-system -value: 1000 \ No newline at end of file diff --git a/manifests/priorityclass/k8s-supply-prd-ase1/priorityclass-high.yaml b/manifests/priorityclass/k8s-supply-prd-ase1/priorityclass-high.yaml deleted file mode 100644 index 038ae8e..0000000 --- a/manifests/priorityclass/k8s-supply-prd-ase1/priorityclass-high.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is high priority and should be used for app service pods only. -kind: PriorityClass -metadata: - name: high-priority - namespace: kube-system -value: 1000000 diff --git a/manifests/priorityclass/k8s-supply-prd-ase1/priorityclass-low.yaml b/manifests/priorityclass/k8s-supply-prd-ase1/priorityclass-low.yaml deleted file mode 100644 index d3446cb..0000000 --- a/manifests/priorityclass/k8s-supply-prd-ase1/priorityclass-low.yaml +++ /dev/null @@ -1,8 +0,0 @@ -apiVersion: scheduling.k8s.io/v1 -description: >- - This priority class is low priority and should be used for non-app pods only. -kind: PriorityClass -metadata: - name: low-priority - namespace: kube-system -value: 1000 \ No newline at end of file diff --git a/manifests/storageclass/gke-central-prd-ase1a/hyperdisk-balanced.yaml b/manifests/storageclass/gke-central-prd-ase1a/hyperdisk-balanced.yaml deleted file mode 100644 index 6ffacdd..0000000 --- a/manifests/storageclass/gke-central-prd-ase1a/hyperdisk-balanced.yaml +++ /dev/null @@ -1,10 +0,0 @@ -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - name: hyperdisk-balanced -parameters: - type: hyperdisk-balanced -provisioner: pd.csi.storage.gke.io -reclaimPolicy: Retain -volumeBindingMode: WaitForFirstConsumer -allowVolumeExpansion: true diff --git a/manifests/storageclass/gke-central-prd-ase1a/pd-standard-retain.yaml b/manifests/storageclass/gke-central-prd-ase1a/pd-standard-retain.yaml deleted file mode 100644 index 0f4f5c2..0000000 --- a/manifests/storageclass/gke-central-prd-ase1a/pd-standard-retain.yaml +++ /dev/null @@ -1,10 +0,0 @@ -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - name: pd-standard-retain -parameters: - type: pd-standard -provisioner: kubernetes.io/gce-pd -reclaimPolicy: Retain -volumeBindingMode: WaitForFirstConsumer -allowVolumeExpansion: true diff --git a/manifests/storageclass/gke-central-prd-ase1a/pulse-nfs-sc-prd.yaml b/manifests/storageclass/gke-central-prd-ase1a/pulse-nfs-sc-prd.yaml deleted file mode 100644 index ede451e..0000000 --- a/manifests/storageclass/gke-central-prd-ase1a/pulse-nfs-sc-prd.yaml +++ /dev/null @@ -1,10 +0,0 @@ -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - name: pulse-nfs-sc-prd -# Statically provisioned NFS storage. PVs are pre-created pointing at the Pulse -# NFS server; this class only binds them (no dynamic provisioner). -provisioner: kubernetes.io/no-provisioner -reclaimPolicy: Delete -volumeBindingMode: Immediate -allowVolumeExpansion: false diff --git a/manifests/storageclass/gke-central-prd-ase1a/sc-pd-ssd.yaml b/manifests/storageclass/gke-central-prd-ase1a/sc-pd-ssd.yaml deleted file mode 100644 index 78bed41..0000000 --- a/manifests/storageclass/gke-central-prd-ase1a/sc-pd-ssd.yaml +++ /dev/null @@ -1,12 +0,0 @@ -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - annotations: - storageclass.kubernetes.io/is-default-class: "true" - name: sc-pd-ssd -parameters: - type: pd-ssd -provisioner: kubernetes.io/gce-pd -reclaimPolicy: Retain -volumeBindingMode: WaitForFirstConsumer -allowVolumeExpansion: true diff --git a/manifests/storageclass/gke-central-prd-ase1a/sc-pd-standard.yaml b/manifests/storageclass/gke-central-prd-ase1a/sc-pd-standard.yaml deleted file mode 100644 index a44479d..0000000 --- a/manifests/storageclass/gke-central-prd-ase1a/sc-pd-standard.yaml +++ /dev/null @@ -1,12 +0,0 @@ -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - annotations: - storageclass.kubernetes.io/is-default-class: "true" - name: sc-pd-standard -parameters: - type: pd-standard -provisioner: kubernetes.io/gce-pd -reclaimPolicy: Retain -volumeBindingMode: WaitForFirstConsumer -allowVolumeExpansion: true diff --git a/manifests/storageclass/gke-datascience-prd-ase1a/enterprise-multishare-rwx.yaml b/manifests/storageclass/gke-datascience-prd-ase1a/enterprise-multishare-rwx.yaml deleted file mode 100644 index c894041..0000000 --- a/manifests/storageclass/gke-datascience-prd-ase1a/enterprise-multishare-rwx.yaml +++ /dev/null @@ -1,20 +0,0 @@ ---- -allowVolumeExpansion: true -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - annotations: - components.gke.io/component-name: filestorecsi - components.gke.io/component-version: 0.10.15 - components.gke.io/layer: addon - labels: - addonmanager.kubernetes.io/mode: EnsureExists - k8s-app: gcp-filestore-csi-driver - name: enterprise-multishare-rwx -parameters: - instance-storageclass-label: enterprise-multishare-rwx - multishare: 'true' - tier: enterprise -provisioner: filestore.csi.storage.gke.io -reclaimPolicy: Delete -volumeBindingMode: WaitForFirstConsumer diff --git a/manifests/storageclass/gke-datascience-prd-ase1a/enterprise-rwx.yaml b/manifests/storageclass/gke-datascience-prd-ase1a/enterprise-rwx.yaml deleted file mode 100644 index e9528fa..0000000 --- a/manifests/storageclass/gke-datascience-prd-ase1a/enterprise-rwx.yaml +++ /dev/null @@ -1,18 +0,0 @@ ---- -allowVolumeExpansion: true -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - annotations: - components.gke.io/component-name: filestorecsi - components.gke.io/component-version: 0.10.15 - components.gke.io/layer: addon - labels: - addonmanager.kubernetes.io/mode: EnsureExists - k8s-app: gcp-filestore-csi-driver - name: enterprise-rwx -parameters: - tier: enterprise -provisioner: filestore.csi.storage.gke.io -reclaimPolicy: Delete -volumeBindingMode: WaitForFirstConsumer diff --git a/manifests/storageclass/gke-datascience-prd-ase1a/hyperdisk-balanced.yaml b/manifests/storageclass/gke-datascience-prd-ase1a/hyperdisk-balanced.yaml deleted file mode 100644 index 4c5ec44..0000000 --- a/manifests/storageclass/gke-datascience-prd-ase1a/hyperdisk-balanced.yaml +++ /dev/null @@ -1,18 +0,0 @@ ---- -allowVolumeExpansion: true -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - annotations: - components.gke.io/component-name: pdcsi - components.gke.io/component-version: 0.13.17 - components.gke.io/layer: addon - labels: - addonmanager.kubernetes.io/mode: EnsureExists - k8s-app: gcp-compute-persistent-disk-csi-driver - name: hyperdisk-balanced -parameters: - type: hyperdisk-balanced -provisioner: pd.csi.storage.gke.io -reclaimPolicy: Retain -volumeBindingMode: WaitForFirstConsumer diff --git a/manifests/storageclass/gke-datascience-prd-ase1a/pd-standard-retain.yaml b/manifests/storageclass/gke-datascience-prd-ase1a/pd-standard-retain.yaml deleted file mode 100644 index 423fb91..0000000 --- a/manifests/storageclass/gke-datascience-prd-ase1a/pd-standard-retain.yaml +++ /dev/null @@ -1,11 +0,0 @@ ---- -allowVolumeExpansion: true -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - name: pd-standard-retain -parameters: - type: pd-standard -provisioner: kubernetes.io/gce-pd -reclaimPolicy: Retain -volumeBindingMode: WaitForFirstConsumer diff --git a/manifests/storageclass/gke-datascience-prd-ase1a/premium-rwx.yaml b/manifests/storageclass/gke-datascience-prd-ase1a/premium-rwx.yaml deleted file mode 100644 index f257f5e..0000000 --- a/manifests/storageclass/gke-datascience-prd-ase1a/premium-rwx.yaml +++ /dev/null @@ -1,18 +0,0 @@ ---- -allowVolumeExpansion: true -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - annotations: - components.gke.io/component-name: filestorecsi - components.gke.io/component-version: 0.10.15 - components.gke.io/layer: addon - labels: - addonmanager.kubernetes.io/mode: EnsureExists - k8s-app: gcp-filestore-csi-driver - name: premium-rwx -parameters: - tier: premium -provisioner: filestore.csi.storage.gke.io -reclaimPolicy: Delete -volumeBindingMode: WaitForFirstConsumer diff --git a/manifests/storageclass/gke-datascience-prd-ase1a/pulse-nfs-sc-prd.yaml b/manifests/storageclass/gke-datascience-prd-ase1a/pulse-nfs-sc-prd.yaml deleted file mode 100644 index 1f28c14..0000000 --- a/manifests/storageclass/gke-datascience-prd-ase1a/pulse-nfs-sc-prd.yaml +++ /dev/null @@ -1,9 +0,0 @@ ---- -allowVolumeExpansion: false -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - name: pulse-nfs-sc-prd -provisioner: kubernetes.io/no-provisioner -reclaimPolicy: Delete -volumeBindingMode: Immediate diff --git a/manifests/storageclass/gke-datascience-prd-ase1a/pulse-nfs-sc-secured-prd.yaml b/manifests/storageclass/gke-datascience-prd-ase1a/pulse-nfs-sc-secured-prd.yaml deleted file mode 100644 index 72fd6fb..0000000 --- a/manifests/storageclass/gke-datascience-prd-ase1a/pulse-nfs-sc-secured-prd.yaml +++ /dev/null @@ -1,9 +0,0 @@ ---- -allowVolumeExpansion: false -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - name: pulse-nfs-sc-secured-prd -provisioner: kubernetes.io/no-provisioner -reclaimPolicy: Delete -volumeBindingMode: Immediate diff --git a/manifests/storageclass/gke-datascience-prd-ase1a/regional-rwx.yaml b/manifests/storageclass/gke-datascience-prd-ase1a/regional-rwx.yaml deleted file mode 100644 index be9375d..0000000 --- a/manifests/storageclass/gke-datascience-prd-ase1a/regional-rwx.yaml +++ /dev/null @@ -1,18 +0,0 @@ ---- -allowVolumeExpansion: true -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - annotations: - components.gke.io/component-name: filestorecsi - components.gke.io/component-version: 0.18.17 - components.gke.io/layer: addon - labels: - addonmanager.kubernetes.io/mode: EnsureExists - k8s-app: gcp-filestore-csi-driver - name: regional-rwx -parameters: - tier: regional -provisioner: filestore.csi.storage.gke.io -reclaimPolicy: Delete -volumeBindingMode: WaitForFirstConsumer diff --git a/manifests/storageclass/gke-datascience-prd-ase1a/sc-filestore-standard.yaml b/manifests/storageclass/gke-datascience-prd-ase1a/sc-filestore-standard.yaml deleted file mode 100644 index a76019c..0000000 --- a/manifests/storageclass/gke-datascience-prd-ase1a/sc-filestore-standard.yaml +++ /dev/null @@ -1,12 +0,0 @@ ---- -allowVolumeExpansion: true -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - name: sc-filestore-standard -parameters: - network: default - tier: standard -provisioner: filestore.csi.storage.gke.io -reclaimPolicy: Retain -volumeBindingMode: WaitForFirstConsumer diff --git a/manifests/storageclass/gke-datascience-prd-ase1a/standard-rwx-retain.yaml b/manifests/storageclass/gke-datascience-prd-ase1a/standard-rwx-retain.yaml deleted file mode 100644 index f07b3e6..0000000 --- a/manifests/storageclass/gke-datascience-prd-ase1a/standard-rwx-retain.yaml +++ /dev/null @@ -1,18 +0,0 @@ ---- -allowVolumeExpansion: true -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - annotations: - components.gke.io/component-name: filestorecsi - components.gke.io/component-version: 0.10.15 - components.gke.io/layer: addon - labels: - addonmanager.kubernetes.io/mode: EnsureExists - k8s-app: gcp-filestore-csi-driver - name: standard-rwx-retain -parameters: - tier: standard -provisioner: filestore.csi.storage.gke.io -reclaimPolicy: Retain -volumeBindingMode: WaitForFirstConsumer diff --git a/manifests/storageclass/gke-datascience-prd-ase1a/standard-rwx.yaml b/manifests/storageclass/gke-datascience-prd-ase1a/standard-rwx.yaml deleted file mode 100644 index 22dd748..0000000 --- a/manifests/storageclass/gke-datascience-prd-ase1a/standard-rwx.yaml +++ /dev/null @@ -1,18 +0,0 @@ ---- -allowVolumeExpansion: true -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - annotations: - components.gke.io/component-name: filestorecsi - components.gke.io/component-version: 0.10.15 - components.gke.io/layer: addon - labels: - addonmanager.kubernetes.io/mode: EnsureExists - k8s-app: gcp-filestore-csi-driver - name: standard-rwx -parameters: - tier: standard -provisioner: filestore.csi.storage.gke.io -reclaimPolicy: Delete -volumeBindingMode: WaitForFirstConsumer diff --git a/manifests/storageclass/gke-datascience-prd-ase1a/zonal-rwx.yaml b/manifests/storageclass/gke-datascience-prd-ase1a/zonal-rwx.yaml deleted file mode 100644 index 10b98d1..0000000 --- a/manifests/storageclass/gke-datascience-prd-ase1a/zonal-rwx.yaml +++ /dev/null @@ -1,18 +0,0 @@ ---- -allowVolumeExpansion: true -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - annotations: - components.gke.io/component-name: filestorecsi - components.gke.io/component-version: 0.10.15 - components.gke.io/layer: addon - labels: - addonmanager.kubernetes.io/mode: EnsureExists - k8s-app: gcp-filestore-csi-driver - name: zonal-rwx -parameters: - tier: zonal -provisioner: filestore.csi.storage.gke.io -reclaimPolicy: Delete -volumeBindingMode: WaitForFirstConsumer diff --git a/post-commit-scripts/._commit-metric.sh b/post-commit-scripts/._commit-metric.sh deleted file mode 100644 index f79ec62787c9ce61d07591a146666a775513ad85..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 311 zcmZQz6=P>$Vqox1Ojhs@R)|o50+1L3ClDJkFfg(LX&|4`97q!Z9795aAj-fxwgB19 zXxc!ggTy@;82FR(bM+Dn3UX5QaubttAPWBgDQgA>QG{G!X<|`gUP)$NDg(p8JL@zo zCd}|!GtXOQj?J!{RS-?2l7;#P1{OxvW|m2bW+?_K*1$Vqox1Ojhs@R)|o50+1L3ClDI}@gg7w@vi_e5x_AdBnYYuq+J^qI7A5ADWagzZ6zUroSQuKHStcc#S(qkSJ7*N-=cZblyO_DS lShzasTDrKH>6$pZnd(|P8yV?3nHsnlxVXAn7+N?o0055kAH@Iw diff --git a/post-commit-scripts/commit-metric.sh b/post-commit-scripts/commit-metric.sh deleted file mode 100644 index 0689961..0000000 --- a/post-commit-scripts/commit-metric.sh +++ /dev/null @@ -1,1551 +0,0 @@ -#!/usr/bin/env bash -# commit-metric.sh — Cursor AI Commit Metric Collector (Bash port of commit-metric.go) -# -# Architecture: Two-phase execution triggered by post-commit git hook. -# -# Phase 1 ("start") — Runs synchronously in the hook (fast): -# 1. Checks if this is a normal commit (skips rebase/merge). -# 2. Gets the latest commit hash from git. -# 3. Writes commit info to a temp file. -# 4. Spawns itself as "continue" in a detached background process. -# -# Phase 2 ("continue") — Runs in background: -# 1. Polls Cursor DB until commit hash appears (500ms interval, 3 min max). -# 2. On match: builds request payload, sends to API. -# 3. On timeout: calls /error endpoint with commit hash. -# 4. Uploads dangling prompt metrics for repos matching the commit. -# 5. Retries failed requests from previous runs. -# -# Dependencies: bash 4+, sqlite3, jq, curl, git - -set -euo pipefail - -# ============================================================ -# Dependency check -# ============================================================ - -check_dependencies() { - mkdir -p ~/bin - - # Detect CPU architecture (Intel vs Apple Silicon) - ARCH=$(uname -m) - if [[ "$ARCH" == "arm64" ]]; then - JQ_URL="https://github.com/stedolan/jq/releases/latest/download/jq-macos-arm64" - elif [[ "$ARCH" == "x86_64" ]]; then - JQ_URL="https://github.com/stedolan/jq/releases/latest/download/jq-osx-amd64" - else - log_warn "Unsupported architecture: $ARCH" - return 1 - fi - - # Download jq if not installed - if ! command -v jq >/dev/null 2>&1; then - curl -fsSL -o ~/bin/jq "$JQ_URL" - chmod +x ~/bin/jq - fi - - # Add to PATH - export PATH="$HOME/bin:$PATH" - - # These should always be present on macOS/Linux - local missing=() - for cmd in sqlite3 curl git; do - if ! command -v "$cmd" >/dev/null 2>&1; then - missing+=("$cmd") - fi - done - if [ ${#missing[@]} -gt 0 ]; then - log_warn "Missing required dependencies: ${missing[*]}" - return 1 - fi -} -# check_dependencies is called from main after parsing the command - -# ============================================================ -# Configuration -# ============================================================ - -# API endpoint to send commit metrics -API_ENDPOINT="https://cursor-server.meeshogcp.in/api/v1/add-commit-metrics" - -# Error API endpoint (called on polling timeout) -ERROR_API_ENDPOINT="https://cursor-server.meeshogcp.in/api/v1/error" - -# Set to true to skip API call and only save locally (for testing) -DRY_RUN=false - -# Database configuration -DB_RELATIVE_PATH="Library/Application Support/Cursor/User/globalStorage/state.vscdb" -TABLE_NAME="ItemTable" -KEY_NAME="aiCodeTracking.recentCommit" - -# SQLite configuration -BUSY_TIMEOUT_MS=3000 -MAX_RETRIES=3 -INITIAL_RETRY_DELAY_MS=500 -MAX_RETRY_DELAY_MS=2000 - -# API retry configuration -API_MAX_ATTEMPTS=3 -API_INITIAL_RETRY_DELAY_MS=1000 -API_MAX_RETRY_DELAY_MS=5000 - -# DB polling configuration (post-commit: wait for Cursor to update DB) -COMMIT_POLL_INTERVAL_MS=10000 -COMMIT_MAX_WAIT_S=120 - -# Storage paths (relative to $HOME) -FAILED_COMMITS_FILE=".cursor-metrics/commit-metric/failed.json" -METRICS_OUTPUT_DIR=".cursor-metrics/commit-metric/data" -TEMP_DIR=".cursor-metrics/commit-metric/tmp" -LOG_DIR_RELATIVE=".cursor-metrics/commit-metric/logs" -REBASE_MAP_FILE=".cursor-metrics/commit-metric/rebase-map.json" - -# ---- Dangling prompt metrics configuration ---- -PROMPT_API_ENDPOINT="https://cursor-server.meeshogcp.in/api/v1/add-prompt-metrics" -PROMPT_DB_TABLE="cursorDiskKV" -PROMPT_PERSISTENT_STORAGE_DIR=".cursor-metrics/prompt-metric/composer-partialDiffFates" -PROMPT_FAILED_REQUESTS_FILE=".cursor-metrics/prompt-metric/failed.json" - -# ============================================================ -# Logging -# ============================================================ - -LOG_FILE="" - -setup_logging() { - local log_dir="${HOME}/${LOG_DIR_RELATIVE}" - mkdir -p "$log_dir" 2>/dev/null || true - LOG_FILE="${log_dir}/commit-metric.log" -} - -# logWarn writes a timestamped warning to the log file with [commit-metric] prefix. -# Falls back to stderr if the log file is not available. -log_warn() { - local fmt_str="$1"; shift - local msg - # shellcheck disable=SC2059 - msg=$(printf "$fmt_str" "$@") - local line - line="[commit-metric] $(date -u +"%Y-%m-%dT%H:%M:%S%z") ${msg}" - if [ -n "$LOG_FILE" ]; then - echo "$line" >> "$LOG_FILE" 2>/dev/null || echo "$line" >&2 - else - echo "$line" >&2 - fi -} - -# ============================================================ -# Utility functions -# ============================================================ - -# min_val returns the smaller of two integers -min_val() { - local a=$1 b=$2 - if [ "$a" -lt "$b" ]; then echo "$a"; else echo "$b"; fi -} - -# sleep_ms sleeps for N milliseconds -sleep_ms() { - local ms=$1 - local secs - secs=$(awk "BEGIN { printf \"%.3f\", $ms / 1000 }") - sleep "$secs" -} - -# get_db_path returns the full path to the Cursor state database -get_db_path() { - echo "${HOME}/${DB_RELATIVE_PATH}" -} - -# ============================================================ -# Value decoding -# ============================================================ - -# is_hex_string checks if a string is hex-encoded (even length, only hex chars) -is_hex_string() { - local s="$1" - local len=${#s} - if [ "$len" -eq 0 ] || [ $(( len % 2 )) -ne 0 ]; then - return 1 - fi - # Check all characters are hex - if [[ "$s" =~ ^[0-9a-fA-F]+$ ]]; then - return 0 - fi - return 1 -} - -# hex_decode reads hex from stdin and outputs raw bytes. -# Uses xxd, perl, or python3 (whichever is available). -hex_decode() { - if command -v xxd >/dev/null 2>&1; then - xxd -r -p - elif command -v perl >/dev/null 2>&1; then - perl -pe 's/(..)/chr(hex($1))/ge' - elif command -v python3 >/dev/null 2>&1; then - python3 -c "import sys,binascii; sys.stdout.buffer.write(binascii.unhexlify(sys.stdin.read().strip()))" - else - return 1 - fi -} - -# decode_value decodes a raw DB value (could be JSON or hex-encoded) to JSON bytes. -# Mirrors Go's decodeValue function. -decode_value() { - local raw="$1" - if [ -z "$raw" ]; then - return 1 - fi - # Check if valid JSON - if echo "$raw" | jq empty 2>/dev/null; then - echo "$raw" - return 0 - fi - # Check if hex-encoded - if is_hex_string "$raw"; then - local decoded - if decoded=$(echo "$raw" | hex_decode 2>/dev/null) && [ -n "$decoded" ]; then - # Verify it's valid JSON (UTF-8 check implicit) - if echo "$decoded" | jq empty 2>/dev/null; then - echo "$decoded" - return 0 - fi - fi - fi - return 1 -} - -# ============================================================ -# DB helpers -# ============================================================ - -# read_db_value reads a single value from the DB by exact key. -# Uses typeof() to handle BLOB values safely (returns hex for BLOBs). -# PRAGMAs output is suppressed via .output /dev/null so it doesn't mix with query results. -read_db_value() { - local db_path="$1" - local table="$2" - local key="$3" - - sqlite3 -readonly "$db_path" 2>/dev/null </dev/null || echo "$(date +%s)000000000") - local file_name="${commit_hash}_${timestamp_ns}.json" - local file_path="${dir}/${file_name}" - - echo "$commit_data" > "$file_path" - echo "$file_path" -} - -# ============================================================ -# Git helpers -# ============================================================ - -# getGitEmail retrieves the user's email from git config -get_git_email() { - local email - email=$(git config --get user.email 2>/dev/null || true) - if [ -z "$email" ]; then - email=$(git config --global --get user.email 2>/dev/null || true) - fi - echo "$email" -} - -# epochMsToUTCString converts epoch milliseconds to UTC string with ms precision. -# Output format: "2006-01-02T15:04:05.000Z" -epoch_ms_to_utc_string() { - local epoch_ms="$1" - if [ -z "$epoch_ms" ] || [ "$epoch_ms" = "0" ] || [ "$epoch_ms" = "null" ]; then - date -u +"%Y-%m-%dT%H:%M:%S.000Z" - return - fi - - local seconds=$(( epoch_ms / 1000 )) - local millis=$(( epoch_ms % 1000 )) - local millis_padded - millis_padded=$(printf "%03d" "$millis") - - local formatted - # GNU date - if formatted=$(date -u -d "@${seconds}" +"%Y-%m-%dT%H:%M:%S" 2>/dev/null); then - echo "${formatted}.${millis_padded}Z" - # BSD/macOS date - elif formatted=$(date -u -r "${seconds}" +"%Y-%m-%dT%H:%M:%S" 2>/dev/null); then - echo "${formatted}.${millis_padded}Z" - else - date -u +"%Y-%m-%dT%H:%M:%S.000Z" - fi -} - -# toUTCString parses any time string (e.g., git's ISO 8601 with timezone) and -# converts it to UTC with ms precision. Returns empty string on parse failure. -to_utc_string() { - local ts="$1" - if [ -z "$ts" ]; then - echo "" - return - fi - - local parsed - # GNU date: handles "+05:30" colon timezone natively - if parsed=$(date -u -d "$ts" +"%Y-%m-%dT%H:%M:%S.000Z" 2>/dev/null); then - echo "$parsed" - return - fi - - # BSD/macOS date: %z expects "+0530" not "+05:30", so strip the colon - # from the timezone offset before parsing. - # "2026-02-15T01:26:12+05:30" -> "2026-02-15T01:26:12+0530" - local ts_nocolon="$ts" - if [[ "$ts" =~ ^(.+)([+-][0-9]{2}):([0-9]{2})$ ]]; then - ts_nocolon="${BASH_REMATCH[1]}${BASH_REMATCH[2]}${BASH_REMATCH[3]}" - fi - if parsed=$(date -u -jf "%Y-%m-%dT%H:%M:%S%z" "$ts_nocolon" +"%Y-%m-%dT%H:%M:%S.000Z" 2>/dev/null); then - echo "$parsed" - return - fi - - # Return as-is if unparseable - echo "$ts" -} - -# is_normal_commit returns 0 for normal commits, 1 for rebase/merge/cherry-pick. -# For rebase and cherry-pick, records the original→replayed hash mapping before skipping. -is_normal_commit() { - local git_dir - git_dir=$(git rev-parse --git-dir 2>/dev/null) || return 1 - - # Skip during rebase (interactive or non-interactive) - if [ -d "${git_dir}/rebase-merge" ] || [ -d "${git_dir}/rebase-apply" ]; then - local original_hash="" - if [ -d "${git_dir}/rebase-merge" ] && [ -f "${git_dir}/rebase-merge/done" ]; then - original_hash=$(tail -1 "${git_dir}/rebase-merge/done" 2>/dev/null | awk '{print $2}') - fi - if [ -z "$original_hash" ] && [ -f "${git_dir}/rebase-apply/original-commit" ]; then - original_hash=$(cat "${git_dir}/rebase-apply/original-commit" 2>/dev/null | tr -d '[:space:]') - fi - [ -n "$original_hash" ] && record_commit_hash_mapping "$original_hash" - return 1 - fi - - # Skip during cherry-pick (CHERRY_PICK_HEAD exists until post-commit cleanup) - if [ -f "${git_dir}/CHERRY_PICK_HEAD" ]; then - local original_hash - original_hash=$(cat "${git_dir}/CHERRY_PICK_HEAD" 2>/dev/null | tr -d '[:space:]') - [ -n "$original_hash" ] && record_commit_hash_mapping "$original_hash" - return 1 - fi - - # Skip merge commits (HEAD has more than 1 parent) - if git rev-parse HEAD^2 >/dev/null 2>&1; then - return 1 - fi - - return 0 -} - -# record_commit_hash_mapping saves replayed_hash→original_hash mapping. -# Used by rebase and cherry-pick to track which original commit was replayed. -# Stored at ~/ as a JSON object keyed by replayed hash. -record_commit_hash_mapping() { - local original_hash="$1" - - # Resolve short hash to full hash - local full_hash - full_hash=$(git rev-parse "$original_hash" 2>/dev/null) || full_hash="$original_hash" - original_hash="$full_hash" - - local replayed_hash - replayed_hash=$(git rev-parse HEAD 2>/dev/null) || return 0 - - local repo_path - repo_path=$(git rev-parse --show-toplevel 2>/dev/null) || return 0 - local repo_name - repo_name=$(get_repo_name_from_path "$repo_path") - local git_dir - git_dir=$(git -C "$repo_path" rev-parse --git-dir 2>/dev/null) || return 0 - local branch_name="" - if [ -f "${git_dir}/rebase-merge/head-name" ]; then - branch_name=$(cat "${git_dir}/rebase-merge/head-name" 2>/dev/null | sed 's|^refs/heads/||') - elif [ -f "${git_dir}/rebase-apply/head-name" ]; then - branch_name=$(cat "${git_dir}/rebase-apply/head-name" 2>/dev/null | sed 's|^refs/heads/||') - fi - if [ -z "$branch_name" ]; then - branch_name=$(git -C "$repo_path" rev-parse --abbrev-ref HEAD 2>/dev/null || true) - fi - - local map_file="${HOME}/${REBASE_MAP_FILE}" - mkdir -p "$(dirname "$map_file")" 2>/dev/null || true - - local current_map="{}" - if [ -f "$map_file" ]; then - current_map=$(cat "$map_file" 2>/dev/null) || current_map="{}" - if ! echo "$current_map" | jq empty 2>/dev/null; then - current_map="{}" - fi - fi - - current_map=$(echo "$current_map" | jq \ - --arg replayed "$replayed_hash" \ - --arg orig "$original_hash" \ - --arg repo "$repo_name" \ - --arg branch "$branch_name" \ - '. + {($replayed): {original: $orig, repo: $repo, branch: $branch}}') - - echo "$current_map" | jq '.' > "$map_file" 2>/dev/null || true - - log_warn "commit mapping recorded: %s → %s (%s)" "$replayed_hash" "$original_hash" "$repo_name" -} - -# getRepoNameFromPath tries git remote origin URL first, falls back to basename. -get_repo_name_from_path() { - local root_path="$1" - - local url - url=$(git -C "$root_path" remote get-url origin 2>/dev/null || true) - if [ -n "$url" ]; then - local name - name=$(parse_repo_name_from_url "$url") - if [ -n "$name" ]; then - echo "$name" - return - fi - fi - - basename "$root_path" -} - -# parseRepoNameFromURL extracts "org/repo" from a git remote URL. -parse_repo_name_from_url() { - local raw_url="$1" - - # SSH: git@github.com:org/repo.git - if [[ "$raw_url" == git@* ]]; then - local after_colon="${raw_url#*:}" - after_colon="${after_colon%.git}" - echo "$after_colon" - return - fi - - # HTTPS: https://github.com/org/repo.git - raw_url="${raw_url%.git}" - local second_last last - last=$(basename "$raw_url") - second_last=$(basename "$(dirname "$raw_url")") - if [ -n "$second_last" ] && [ -n "$last" ]; then - echo "${second_last}/${last}" - return - fi -} - -# ============================================================ -# Repo path resolution via Cursor's repositoryTracker.paths -# ============================================================ - -# resolve_repo_local_path finds the local filesystem path for a repo name -# by searching Cursor's repositoryTracker.paths. -# -# Matching: CursorCommitData.RepoName (e.g. "meesho/cursor-metrics-instrumentation") -# is matched case-insensitively against tracker keys (e.g. "github.com/meesho/cursor-metrics-instrumentation") -# using suffix matching. -resolve_repo_local_path() { - local repo_name="$1" - local tracker_paths_json="$2" - - if [ -z "$repo_name" ] || [ "$tracker_paths_json" = "{}" ] || [ -z "$tracker_paths_json" ]; then - echo "" - return - fi - - local repo_name_lower - repo_name_lower=$(echo "$repo_name" | tr '[:upper:]' '[:lower:]') - - # Iterate tracker paths keys and find suffix match - local result - result=$(echo "$tracker_paths_json" | jq -r --arg rn "$repo_name_lower" ' - to_entries[] | - select( - (.key | ascii_downcase) as $k | - ($k | endswith("/" + $rn)) or ($k == $rn) - ) | .value.localPath // empty - ' 2>/dev/null | head -1) - - if [ -n "$result" ]; then - # Remove file:// prefix - echo "${result#file://}" - fi -} - -# ============================================================ -# Convert to request -# ============================================================ - -# convertToRequest converts CursorDB data to the server request format. -# repo_path is passed directly from run_continue (known from the post-commit hook). -convert_to_request() { - local commit_data_json="$1" - local repo_path="$2" - - # Get user email from git config - local email - email=$(get_git_email) - if [ -z "$email" ]; then - log_warn "could not determine git user email" - return 1 - fi - - # Extract fields from commit data - local commit_hash repo_name branch_name - local tab_lines_added tab_lines_deleted composer_lines_added composer_lines_deleted - local lines_added lines_deleted - - commit_hash=$(echo "$commit_data_json" | jq -r '.commitHash // ""') - repo_name=$(echo "$commit_data_json" | jq -r '.repoName // ""') - branch_name=$(echo "$commit_data_json" | jq -r '.branchName // ""') - tab_lines_added=$(echo "$commit_data_json" | jq -r '.tabLinesAdded // 0') - tab_lines_deleted=$(echo "$commit_data_json" | jq -r '.tabLinesDeleted // 0') - composer_lines_added=$(echo "$commit_data_json" | jq -r '.composerLinesAdded // 0') - composer_lines_deleted=$(echo "$commit_data_json" | jq -r '.composerLinesDeleted // 0') - lines_added=$(echo "$commit_data_json" | jq -r '.linesAdded // 0') - lines_deleted=$(echo "$commit_data_json" | jq -r '.linesDeleted // 0') - - # Get commit timestamp from git - local timestamp_str="" - if [ -n "$commit_hash" ]; then - local git_ts - git_ts=$(git -C "$repo_path" log -1 --format="%aI" "$commit_hash" 2>/dev/null || true) - if [ -n "$git_ts" ]; then - timestamp_str=$(to_utc_string "$git_ts") - fi - fi - - # Get parent commit timestamp from git - local parent_timestamp="" - if [ -n "$commit_hash" ]; then - local parent_ts - parent_ts=$(git -C "$repo_path" log -1 --format="%aI" "${commit_hash}~1" 2>/dev/null || true) - if [ -n "$parent_ts" ]; then - parent_timestamp=$(to_utc_string "$parent_ts") - else - parent_timestamp="$timestamp_str" - fi - fi - - # Build request JSON - jq -n \ - --arg email "$email" \ - --arg commit_hash "$commit_hash" \ - --arg timestamp "$timestamp_str" \ - --arg parent_commit_timestamp "$parent_timestamp" \ - --arg repo "$repo_name" \ - --arg branch "$branch_name" \ - --argjson tabLinesAdded "$tab_lines_added" \ - --argjson tabLinesDeleted "$tab_lines_deleted" \ - --argjson composerLinesAdded "$composer_lines_added" \ - --argjson composerLinesDeleted "$composer_lines_deleted" \ - --argjson linesAdded "$lines_added" \ - --argjson linesDeleted "$lines_deleted" \ - '{ - email: $email, - commit_hash: $commit_hash, - timestamp: $timestamp, - parent_commit_timestamp: $parent_commit_timestamp, - repo: $repo, - branch: $branch, - tabLinesAdded: $tabLinesAdded, - tabLinesDeleted: $tabLinesDeleted, - composerLinesAdded: $composerLinesAdded, - composerLinesDeleted: $composerLinesDeleted, - linesAdded: $linesAdded, - linesDeleted: $linesDeleted, - metadata: null - }' -} - -# ============================================================ -# Local metrics storage -# ============================================================ - -# saveMetricsLocally saves the commit metrics to a local JSON file. -# Path: ~//.json -save_metrics_locally() { - local request_json="$1" - - local dir="${HOME}/${METRICS_OUTPUT_DIR}" - mkdir -p "$dir" - - local commit_hash - commit_hash=$(echo "$request_json" | jq -r '.commit_hash // "unknown"') - local file_path="${dir}/${commit_hash}.json" - - # Idempotent — skip if already written - if [ -f "$file_path" ]; then - printf "Metrics already saved locally: %s\n" "$file_path" - return 0 - fi - - echo "$request_json" | jq '.' > "$file_path" - printf "Metrics saved locally: %s\n" "$file_path" -} - -# ============================================================ -# Failed requests persistence -# ============================================================ - -get_failed_commits_path() { - echo "${HOME}/${FAILED_COMMITS_FILE}" -} - -# load_failed_commits reads previously failed commits from the cache file. -# Concurrency is handled by the caller via acquire_lock. -load_failed_commits() { - local path - path=$(get_failed_commits_path) - if [ ! -f "$path" ]; then - echo "[]" - return - fi - - local data - data=$(cat "$path" 2>/dev/null || true) - - if [ -n "$data" ] && echo "$data" | jq empty 2>/dev/null; then - echo "$data" - else - echo "[]" - fi -} - -# saveFailedCommits writes the failed batch to the cache file. -# Pass empty or "[]" to clear the file (on success). -save_failed_commits() { - local commits_json="$1" - local path - path=$(get_failed_commits_path) - - if [ -z "$commits_json" ] || [ "$commits_json" = "[]" ] || [ "$commits_json" = "null" ]; then - rm -f "$path" 2>/dev/null || true - return - fi - - mkdir -p "$(dirname "$path")" 2>/dev/null || true - echo "$commits_json" | jq '.' > "$path" 2>/dev/null || true -} - -# ============================================================ -# Lock helpers -# ============================================================ - -CONTINUE_LOCK_DIR="${HOME}/.cursor-metrics/commit-metric/continue.lock" -DANGLING_LOCK_DIR="${HOME}/.cursor-metrics/prompt-metric/continue.lock" - -acquire_lock() { - local lock_dir="$1" - mkdir -p "$(dirname "$lock_dir")" 2>/dev/null || true - - local poll_ms=500 - local stale_threshold_s=120 - local max_wait_s=180 - local start_time - start_time=$(date +%s) - - while ! mkdir "$lock_dir" 2>/dev/null; do - local now - now=$(date +%s) - - if [ $(( now - start_time )) -gt "$max_wait_s" ]; then - log_warn "lock wait exceeded %ds, force-removing: %s" "$max_wait_s" "$lock_dir" - rmdir "$lock_dir" 2>/dev/null || true - continue - fi - - if [ -d "$lock_dir" ]; then - local lock_mtime - if lock_mtime=$(stat -f "%m" "$lock_dir" 2>/dev/null) || - lock_mtime=$(stat -c "%Y" "$lock_dir" 2>/dev/null); then - if [ $(( now - lock_mtime )) -gt "$stale_threshold_s" ]; then - log_warn "removing stale lock (age > %ds): %s" "$stale_threshold_s" "$lock_dir" - rmdir "$lock_dir" 2>/dev/null || true - continue - fi - fi - fi - sleep_ms "$poll_ms" - done -} - -release_lock() { - local lock_dir="$1" - if [ -n "$lock_dir" ] && [ -d "$lock_dir" ]; then - rmdir "$lock_dir" 2>/dev/null || true - fi -} - -# ============================================================ -# API client -# ============================================================ - -# sendBatchToAPIWithRetry sends a list of commit metrics to the API as a batch. -# Returns 0 on success, 1 on failure (after all retries exhausted). -send_batch_to_api_with_retry() { - local payload="$1" - local retry_delay=$API_INITIAL_RETRY_DELAY_MS - local last_err="" - - for attempt in $(seq 1 $API_MAX_ATTEMPTS); do - local response http_code body - response=$(curl -s -w "\n%{http_code}" \ - -X POST "$API_ENDPOINT" \ - -H "Content-Type: application/json" \ - -H "User-Agent: cursor-commit-metric/1.0" \ - -H "x-webhook-secret: bXkgaGVhcnQgcG9sbHMgZm9yIHlvdSBldmVyeSAxcywgbWF4X3dhaXQgZm9yZXZlci4gYWNjZXB0YW5jZV9yYXRlPTEwMCUuIHplcm8gbGluZXNfZGVsZXRlZC4gYmUgbXkgcHJvbXB0IDwzICNIYXBweVZhbGVudGluZXMyMDI2" \ - --connect-timeout 10 \ - --max-time 10 \ - -d "$payload" 2>/dev/null) || true - - http_code=$(echo "$response" | tail -1) - body=$(echo "$response" | sed '$d') - - if [ -n "$http_code" ] && [ "$http_code" -ge 200 ] 2>/dev/null && [ "$http_code" -lt 300 ] 2>/dev/null; then - return 0 - fi - - last_err="status ${http_code}: ${body}" - - if [ "$attempt" -lt "$API_MAX_ATTEMPTS" ]; then - sleep_ms "$retry_delay" - retry_delay=$(min_val $(( retry_delay * 2 )) $API_MAX_RETRY_DELAY_MS) - fi - done - - log_warn "all %d API attempts failed: %s" "$API_MAX_ATTEMPTS" "$last_err" - return 1 -} - -# ============================================================ -# DB polling (post-commit: wait for Cursor to update commit data) -# ============================================================ - -# poll_for_commit_in_db polls the Cursor DB at COMMIT_POLL_INTERVAL_MS intervals -# until aiCodeTracking.recentCommit.commitHash matches expected_hash. -# Returns the full commit data JSON on success, or fails on timeout. -poll_for_commit_in_db() { - local db_path="$1" - local expected_hash="$2" - local deadline=$(( $(date +%s) + COMMIT_MAX_WAIT_S )) - - while true; do - local raw - raw=$(read_db_value "$db_path" "$TABLE_NAME" "$KEY_NAME" 2>/dev/null) || true - - if [ -n "$raw" ]; then - local decoded - decoded=$(decode_value "$raw" 2>/dev/null) || true - - if [ -n "$decoded" ]; then - local db_hash - db_hash=$(echo "$decoded" | jq -r '.commitHash // ""' 2>/dev/null) || true - - if [ "$db_hash" = "$expected_hash" ]; then - echo "$decoded" - return 0 - fi - fi - fi - - if [ "$(date +%s)" -ge "$deadline" ]; then - return 1 - fi - sleep_ms "$COMMIT_POLL_INTERVAL_MS" - done -} - -# ============================================================ -# Error API (called on polling timeout) -# ============================================================ - -send_error_to_api() { - local commit_hash="$1" - local error_msg="$2" - - local email - email=$(get_git_email) - - local branch="${3:-}" - local repo_name="${4:-}" - - local payload - payload=$(jq -n \ - --arg commit_hash "$commit_hash" \ - --arg email "$email" \ - --arg error "$error_msg" \ - --arg branch "$branch" \ - --arg repo "$repo_name" \ - '{commit_hash: $commit_hash, email: $email, error: $error, branch: $branch, repo: $repo}') - - curl -s -X POST "$ERROR_API_ENDPOINT" \ - -H "Content-Type: application/json" \ - -H "User-Agent: cursor-commit-metric/1.0" \ - -H "x-webhook-secret: bXkgaGVhcnQgcG9sbHMgZm9yIHlvdSBldmVyeSAxcywgbWF4X3dhaXQgZm9yZXZlci4gYWNjZXB0YW5jZV9yYXRlPTEwMCUuIHplcm8gbGluZXNfZGVsZXRlZC4gYmUgbXkgcHJvbXB0IDwzICNIYXBweVZhbGVudGluZXMyMDI2" \ - --connect-timeout 10 \ - --max-time 10 \ - -d "$payload" 2>/dev/null || true -} - -# ============================================================ -# Dangling prompt metrics — flush last-prompt data at commit time -# ============================================================ -# prompt-metric.sh (beforeSubmitPrompt hook) uploads metrics of the -# PREVIOUS prompt. The very last prompt's metrics are therefore -# never uploaded. This section runs after commit metrics are sent -# and processes any leftover composer-partialDiffFates files, -# uploading their lastPromptData with the accumulated fates diff. -# -# When target_repo is provided, only processes composers whose -# .repo field contains the target repo (exact match within || list). - -# ---- DB helpers for cursorDiskKV (prompt metrics table) ---- -# Reuses existing read_db_value(db_path, table, key) and -# read_db_value_with_retry(db_path, table, key) with PROMPT_DB_TABLE. - -query_fates_key_names() { - local db_path="$1" - local composer_id="$2" - local prefix="codeBlockPartialInlineDiffFates:${composer_id}:" - - sqlite3 -readonly "$db_path" 2>/dev/null </dev/null 2>&1; then - sha256sum | cut -d' ' -f1 - elif command -v shasum >/dev/null 2>&1; then - shasum -a 256 | cut -d' ' -f1 - else - openssl dgst -sha256 -hex 2>/dev/null | awk '{print $NF}' - fi -} - -build_range_key() { - local fates_json="$1" - echo "$fates_json" | jq -r ' - [.fates // [] | .[] | - "\(.removedRange.startLineNumber):\(.removedRange.endLineNumberExclusive)::\(.addedRange.endLineNumberExclusive):\(.addedRange.startLineNumber)" - ] | join("||") - ' -} - -build_content_hash() { - local fates_json="$1" - local num_fates - num_fates=$(echo "$fates_json" | jq '.fates | length') - - { - for ((i=0; i/dev/null) || continue - if [ -z "$fates_data" ]; then - continue - fi - - local range_key content_hash composite composite_hash - range_key=$(build_range_key "$fates_data") - content_hash=$(build_content_hash "$fates_data") - composite="${range_key}|CONTENT|${content_hash}" - composite_hash=$(printf '%s' "$composite" | sha256_hash) - - if [ ! -f "${dedup_dir}/${composite_hash}" ]; then - echo "$composite_hash" >> "$order_file" - fi - printf '%s' "$id" > "${dedup_dir}/${composite_hash}" - done <<< "$new_ids_str" - - while IFS= read -r hash; do - cat "${dedup_dir}/${hash}" - echo - done < "$order_file" - - rm -rf "$dedup_dir" -} - -# ---- Failed prompt requests persistence ---- - -get_failed_prompt_requests_path() { - echo "${HOME}/${PROMPT_FAILED_REQUESTS_FILE}" -} - -load_failed_prompt_requests() { - local path - path=$(get_failed_prompt_requests_path) - if [ ! -f "$path" ]; then - echo "[]" - return - fi - - local data - data=$(cat "$path" 2>/dev/null || true) - - if [ -n "$data" ] && echo "$data" | jq empty 2>/dev/null; then - echo "$data" - else - echo "[]" - fi -} - -save_failed_prompt_requests() { - local requests_json="$1" - local path - path=$(get_failed_prompt_requests_path) - - if [ -z "$requests_json" ] || [ "$requests_json" = "[]" ] || [ "$requests_json" = "null" ]; then - rm -f "$path" 2>/dev/null || true - return - fi - - mkdir -p "$(dirname "$path")" 2>/dev/null || true - echo "$requests_json" | jq '.' > "$path" 2>/dev/null || true -} - -# ---- Prompt metrics API sender (uses PROMPT_API_ENDPOINT) ---- - -send_prompt_batch_to_api_with_retry() { - local payload="$1" - local retry_delay=$API_INITIAL_RETRY_DELAY_MS - local last_err="" - - for attempt in $(seq 1 $API_MAX_ATTEMPTS); do - local response http_code body - response=$(curl -s -w "\n%{http_code}" \ - -X POST "$PROMPT_API_ENDPOINT" \ - -H "Content-Type: application/json" \ - -H "User-Agent: cursor-prompt-metric/1.0" \ - -H "x-webhook-secret: bXkgaGVhcnQgcG9sbHMgZm9yIHlvdSBldmVyeSAxcywgbWF4X3dhaXQgZm9yZXZlci4gYWNjZXB0YW5jZV9yYXRlPTEwMCUuIHplcm8gbGluZXNfZGVsZXRlZC4gYmUgbXkgcHJvbXB0IDwzICNIYXBweVZhbGVudGluZXMyMDI2" \ - --connect-timeout 10 \ - --max-time 10 \ - -d "$payload" 2>/dev/null) || true - - http_code=$(echo "$response" | tail -1) - body=$(echo "$response" | sed '$d') - - if [ -n "$http_code" ] && [ "$http_code" -ge 200 ] 2>/dev/null && [ "$http_code" -lt 300 ] 2>/dev/null; then - return 0 - fi - - last_err="status ${http_code}: ${body}" - - if [ "$attempt" -lt "$API_MAX_ATTEMPTS" ]; then - sleep_ms "$retry_delay" - retry_delay=$(min_val $(( retry_delay * 2 )) $API_MAX_RETRY_DELAY_MS) - fi - done - - log_warn "[dangling] all %d prompt API attempts failed: %s" "$API_MAX_ATTEMPTS" "$last_err" - return 1 -} - -# ---- Main dangling upload function ---- - -upload_dangling_prompt_metrics() { - local target_repo="${1:-}" - - # wait for 1 minute to get the unaccepted lines of this commit to get auto accept in db - sleep 60 - - local persistent_dir="${HOME}/${PROMPT_PERSISTENT_STORAGE_DIR}" - - if [ ! -d "$persistent_dir" ]; then - return 0 - fi - - local files=("$persistent_dir"/*.json) - if [ ! -f "${files[0]:-}" ]; then - return 0 - fi - - local db_path - db_path=$(get_db_path) - if [ ! -f "$db_path" ]; then - log_warn "[dangling] database not found at %s" "$db_path" - return 0 - fi - - local user_email - user_email=$(get_git_email) - if [ -z "$user_email" ]; then - log_warn "[dangling] could not determine git user email" - return 0 - fi - - acquire_lock "$DANGLING_LOCK_DIR" - trap 'release_lock "$DANGLING_LOCK_DIR"' EXIT - - local all_requests="[]" - local files_to_delete=() - - for file in "${files[@]}"; do - [ ! -f "$file" ] && continue - - local composer_id - composer_id=$(basename "$file" .json) - - local state - state=$(cat "$file" 2>/dev/null) || continue - if ! echo "$state" | jq empty 2>/dev/null; then - log_warn "[dangling] invalid JSON in %s, skipping" "$file" - files_to_delete+=("$file") - continue - fi - - # If target_repo is specified, only process composers for matching repos. - # The .repo field can be "org/repo" or "org1/repo1||org2/repo2" for multi-root. - if [ -n "$target_repo" ]; then - local file_repo - file_repo=$(echo "$state" | jq -r '.repo // ""') - local delimited_repos="||${file_repo}||" - if [[ "$delimited_repos" != *"||${target_repo}||"* ]]; then - continue - fi - fi - - local last_prompt_data - last_prompt_data=$(echo "$state" | jq '.lastPromptData // {}') - - local prompt_time - prompt_time=$(echo "$last_prompt_data" | jq -r '.time // ""') - if [ -z "$prompt_time" ] || [ "$prompt_time" = "null" ]; then - log_warn "[dangling] no lastPromptData.time for composer %s, skipping" "$composer_id" - files_to_delete+=("$file") - continue - fi - - # ---- Fates diff: find new fates IDs since last upload ---- - local known_fates_ids_json - known_fates_ids_json=$(echo "$state" | jq '.partialInlineDiffFatesIds // []') - - local fates_key_names - fates_key_names=$(query_fates_key_names_with_retry "$db_path" "$composer_id" 2>/dev/null) || true - local all_fates_ids - all_fates_ids=$(extract_fates_ids_from_keys "$fates_key_names" "$composer_id") - - local new_fates_ids="" - if [ -n "$all_fates_ids" ]; then - while IFS= read -r id; do - [ -z "$id" ] && continue - local is_known - is_known=$(echo "$known_fates_ids_json" | jq --arg id "$id" 'any(. == $id)') - if [ "$is_known" = "false" ]; then - if [ -n "$new_fates_ids" ]; then - new_fates_ids="${new_fates_ids}"$'\n'"${id}" - else - new_fates_ids="$id" - fi - fi - done <<< "$all_fates_ids" - fi - - # ---- Read fates data for new IDs ---- - FATES_DATA_DIR=$(mktemp -d) - if [ -n "$new_fates_ids" ]; then - while IFS= read -r id; do - [ -z "$id" ] && continue - local fd - fd=$(read_fates_data "$db_path" "$composer_id" "$id" 2>/dev/null) || { - log_warn "[dangling] fates %s read failed for composer %s" "$id" "$composer_id" - continue - } - if [ -n "$fd" ]; then - echo "$fd" > "${FATES_DATA_DIR}/${id}" - fi - done <<< "$new_fates_ids" - fi - - # ---- Deduplicate ---- - local unique_ids - unique_ids=$(deduplicate_fates "$new_fates_ids") - - # ---- Build chunks + totals ---- - local chunks_json="{}" - local total_sug_added=0 total_sug_removed=0 total_acc_added=0 total_acc_removed=0 - - if [ -n "$unique_ids" ]; then - while IFS= read -r id; do - [ -z "$id" ] && continue - local fd - fd=$(cat "${FATES_DATA_DIR}/${id}" 2>/dev/null) || continue - [ -z "$fd" ] && continue - - local entries_and_totals - entries_and_totals=$(echo "$fd" | jq ' - .fates // [] | reduce .[] as $f ( - { entries: [], sugAdded: 0, sugRemoved: 0, accAdded: 0, accRemoved: 0 }; - ($f.addedRange.endLineNumberExclusive - $f.addedRange.startLineNumber) as $added | - ($f.removedRange.endLineNumberExclusive - $f.removedRange.startLineNumber) as $removed | - .entries += [{ linesAdded: $added, linesRemoved: $removed, fate: $f.fate }] | - .sugAdded += $added | - .sugRemoved += $removed | - (if $f.fate == "accepted" then .accAdded += $added | .accRemoved += $removed else . end) - ) - ') - - local entries - entries=$(echo "$entries_and_totals" | jq '.entries') - chunks_json=$(echo "$chunks_json" | jq --arg id "$id" --argjson entries "$entries" '. + {($id): $entries}') - - total_sug_added=$(( total_sug_added + $(echo "$entries_and_totals" | jq '.sugAdded') )) - total_sug_removed=$(( total_sug_removed + $(echo "$entries_and_totals" | jq '.sugRemoved') )) - total_acc_added=$(( total_acc_added + $(echo "$entries_and_totals" | jq '.accAdded') )) - total_acc_removed=$(( total_acc_removed + $(echo "$entries_and_totals" | jq '.accRemoved') )) - done <<< "$unique_ids" - fi - - [ -n "$FATES_DATA_DIR" ] && rm -rf "$FATES_DATA_DIR" - - # ---- Build request (same shape as prompt-metric.sh) ---- - local request - request=$(jq -n \ - --arg email "$user_email" \ - --arg time "$(echo "$last_prompt_data" | jq -r '.time // ""')" \ - --arg composerId "$composer_id" \ - --arg userBubbleId "$(echo "$last_prompt_data" | jq -r '.userBubbleId // ""')" \ - --arg prompt "$(echo "$last_prompt_data" | jq -r '.prompt // ""')" \ - --argjson isMax "$(echo "$last_prompt_data" | jq '.isMax // false')" \ - --arg mode "$(echo "$last_prompt_data" | jq -r '.mode // ""')" \ - --arg model "$(echo "$last_prompt_data" | jq -r '.model // ""')" \ - --arg repo "$(echo "$state" | jq -r '.repo // ""')" \ - --arg branch "$(echo "$last_prompt_data" | jq -r '.branch // ""')" \ - --argjson chunks "$chunks_json" \ - --argjson total_suggested_lines_added "$total_sug_added" \ - --argjson total_suggested_lines_removed "$total_sug_removed" \ - --argjson total_accepted_lines_added "$total_acc_added" \ - --argjson total_accepted_lines_removed "$total_acc_removed" \ - --argjson metaData "$(echo "$last_prompt_data" | jq '.metadata // null')" \ - '{ - email: $email, - time: $time, - composerId: $composerId, - userBubbleId: $userBubbleId, - prompt: $prompt, - isMax: $isMax, - mode: $mode, - model: $model, - repo: $repo, - branch: $branch, - chunks: $chunks, - total_suggested_lines_added: $total_suggested_lines_added, - total_suggested_lines_removed: $total_suggested_lines_removed, - total_accepted_lines_added: $total_accepted_lines_added, - total_accepted_lines_removed: $total_accepted_lines_removed, - metaData: $metaData - }') - - all_requests=$(echo "$all_requests" | jq --argjson req "$request" '. + [$req]') - files_to_delete+=("$file") - done - - # ---- Send batch ---- - local batch_count - batch_count=$(echo "$all_requests" | jq 'length') - - if [ "$batch_count" -eq 0 ]; then - for f in "${files_to_delete[@]}"; do - rm -f "$f" - done - release_lock "$DANGLING_LOCK_DIR" - return 0 - fi - - if [ "$DRY_RUN" = true ]; then - log_warn "[dangling] dry run: would send %d dangling prompt request(s)" "$batch_count" - local dangling_prompt_metrics_file="${HOME}/.cursor-metrics/prompt-metric/dangling_prompt_metrics.json" - echo "$all_requests" | jq '.' > "$dangling_prompt_metrics_file" - release_lock "$DANGLING_LOCK_DIR" - return 0 - fi - - local previous_failed - previous_failed=$(load_failed_prompt_requests) - - local batch - batch=$(echo "$previous_failed" | jq --argjson reqs "$all_requests" '. + $reqs') - - local total_batch prev_count - total_batch=$(echo "$batch" | jq 'length') - prev_count=$(echo "$previous_failed" | jq 'length') - log_warn "[dangling] sending batch of %d prompt metric(s) (%d dangling + %d previously failed)" \ - "$total_batch" "$batch_count" "$prev_count" - - if send_prompt_batch_to_api_with_retry "$batch"; then - save_failed_prompt_requests "" - log_warn "[dangling] successfully sent %d prompt metric(s)" "$total_batch" - else - log_warn "[dangling] API batch send failed (%d items)" "$total_batch" - save_failed_prompt_requests "$batch" - fi - - for f in "${files_to_delete[@]}"; do - rm -f "$f" - done - - release_lock "$DANGLING_LOCK_DIR" -} - -# ============================================================ -# Phase 1: start — runs synchronously in the post-commit hook (fast) -# ============================================================ - -run_start() { - # Skip non-normal commits (rebase, merge) - if ! is_normal_commit; then - log_warn "skipping non-normal commit (rebase or merge)" - return 0 - fi - - # Get the latest commit hash from git (HEAD is the new commit in post-commit) - local commit_hash - commit_hash=$(git rev-parse HEAD 2>/dev/null) || { - log_warn "failed to get HEAD commit hash" - return 1 - } - - # Get repo path and derive repo name - local repo_path - repo_path=$(git rev-parse --show-toplevel 2>/dev/null) || { - log_warn "failed to get repo toplevel path" - return 1 - } - - local repo_name - repo_name=$(get_repo_name_from_path "$repo_path") - - # Write temp file with commit info for the background process - local temp_data - temp_data=$(jq -n \ - --arg commitHash "$commit_hash" \ - --arg repoName "$repo_name" \ - --arg repoPath "$repo_path" \ - '{commitHash: $commitHash, repoName: $repoName, repoPath: $repoPath}') - - local temp_file_path - temp_file_path=$(write_temp_file "$temp_data") - - # Spawn "continue" as a detached background process - local self_path - self_path=$(realpath "$0" 2>/dev/null || echo "$0") - local continue_log_dir="${HOME}/${LOG_DIR_RELATIVE}" - mkdir -p "$continue_log_dir" 2>/dev/null || true - local continue_log="${continue_log_dir}/continue.log" - - nohup bash "$self_path" continue "$temp_file_path" >/dev/null 2>>"$continue_log" & - disown 2>/dev/null || true -} - -# ============================================================ -# Phase 2: continue — runs in background (slow work) -# ============================================================ - -run_continue() { - local temp_file_path="$1" - - if [ ! -f "$temp_file_path" ]; then - log_warn "temp file not found: %s" "$temp_file_path" - return 1 - fi - - # Read temp file and delete immediately - local temp_data - temp_data=$(cat "$temp_file_path") - rm -f "$temp_file_path" - - if ! echo "$temp_data" | jq empty 2>/dev/null; then - log_warn "parse temp data: invalid JSON" - return 1 - fi - - local commit_hash repo_name repo_path - commit_hash=$(echo "$temp_data" | jq -r '.commitHash') - repo_name=$(echo "$temp_data" | jq -r '.repoName') - repo_path=$(echo "$temp_data" | jq -r '.repoPath') - - local branch_name - branch_name=$(git -C "$repo_path" rev-parse --abbrev-ref HEAD 2>/dev/null || true) - - # Get database path - local db_path - db_path=$(get_db_path) - if [ ! -f "$db_path" ]; then - log_warn "cursor database not found at: %s" "$db_path" - return 1 - fi - - # Poll DB until commit hash matches (aggressive: 500ms interval, 3 min max) - log_warn "polling DB for commit hash %s (max %ds, interval %dms)..." \ - "$commit_hash" "$COMMIT_MAX_WAIT_S" "$COMMIT_POLL_INTERVAL_MS" - - local cursor_data - if ! cursor_data=$(poll_for_commit_in_db "$db_path" "$commit_hash"); then - log_warn "polling timeout: commit hash %s not found in DB within %ds" \ - "$commit_hash" "$COMMIT_MAX_WAIT_S" - send_error_to_api "$commit_hash" "polling_timeout" "$branch_name" "$repo_name" - # Still attempt dangling prompt upload even on timeout - upload_dangling_prompt_metrics "$repo_name" || log_warn "[dangling] upload_dangling_prompt_metrics failed" - return 1 - fi - - log_warn "commit hash %s found in DB, processing..." "$commit_hash" - - # Convert commit data to request (repo_path passed directly for git timestamp lookups) - local request - request=$(convert_to_request "$cursor_data" "$repo_path") - if [ -z "$request" ]; then - log_warn "convert to request failed for commit %s" "$commit_hash" - upload_dangling_prompt_metrics "$repo_name" || log_warn "[dangling] upload_dangling_prompt_metrics failed" - return 1 - fi - - if [ "$DRY_RUN" = true ]; then - save_metrics_locally "$request" - upload_dangling_prompt_metrics "$repo_name" || log_warn "[dangling] upload_dangling_prompt_metrics failed" - return 0 - fi - - # Serialise access to failed.json so concurrent continue processes - # don't overwrite each other's data. - acquire_lock "$CONTINUE_LOCK_DIR" - trap 'release_lock "$CONTINUE_LOCK_DIR"' EXIT - - # Load previously failed commits and merge with current. - local previous_failed - previous_failed=$(load_failed_commits) - - local batch - batch=$(echo "$previous_failed" | jq --argjson req "$request" '. + [$req]') - - local batch_count prev_count - batch_count=$(echo "$batch" | jq 'length') - prev_count=$(echo "$previous_failed" | jq 'length') - - log_warn "Sending batch of %d commit(s) to API (%d previously failed + 1 current)..." \ - "$batch_count" "$prev_count" - - if send_batch_to_api_with_retry "$batch"; then - save_failed_commits "" - log_warn "Successfully sent %d commit(s) to API" "$batch_count" - else - log_warn "API batch send failed (%d items)" "$batch_count" - save_failed_commits "$batch" - fi - - release_lock "$CONTINUE_LOCK_DIR" - - # Flush dangling prompt metrics for repos matching this commit - upload_dangling_prompt_metrics "$repo_name" || log_warn "[dangling] upload_dangling_prompt_metrics failed" - - return 0 -} - -# ============================================================ -# Main -# ============================================================ - -setup_logging - -# Determine the subcommand. Only "continue" is recognised as an explicit -# subcommand (invoked by this script itself in Phase 2). Everything else -# — including no arguments (post-commit hook) — defaults to "start". -CMD="${1:-start}" -if [ "$CMD" != "continue" ]; then - CMD="start" -fi - -# For "start": guarantee exit 0 so the git hook never blocks, -# even if the script crashes, deps are missing, or any error occurs. -if [ "$CMD" = "start" ]; then - trap 'exit 0' EXIT -fi - -case "$CMD" in - start) - check_dependencies || exit 0 - if ! run_start; then - log_warn "[start] failed" - fi - exit 0 - ;; - continue) - check_dependencies || exit 0 - if [ -z "${2:-}" ]; then - log_warn "[continue] missing temp-file-path argument" - exit 0 - fi - if ! run_continue "$2"; then - log_warn "[continue] failed" - exit 0 - fi - ;; -esac \ No newline at end of file diff --git a/post-commit-scripts/runner.sh b/post-commit-scripts/runner.sh deleted file mode 100644 index 78622d8..0000000 --- a/post-commit-scripts/runner.sh +++ /dev/null @@ -1,47 +0,0 @@ -#!/usr/bin/env bash - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" - -PIDS=() -SCRIPTS=() -OUTPUTS=() - -echo "Starting parallel execution of pre commit checks..." - -for script in "$SCRIPT_DIR"/*.sh; do - if [ -x "$script" ] && [ "$(basename "$script")" != "runner.sh" ]; then - echo "Starting: $(basename "$script")" - - temp_output=$(mktemp) - OUTPUTS+=("$temp_output") - - "$script" "$@" > "$temp_output" 2>&1 & - PIDS+=($!) - SCRIPTS+=("$script") - fi -done - -FAILED=0 -for i in "${!PIDS[@]}"; do - if ! wait "${PIDS[$i]}"; then - echo "❌ Failed: $(basename "${SCRIPTS[$i]}")" - echo "Error output:" - echo "----------------------------------------" - cat "${OUTPUTS[$i]}" - echo "----------------------------------------" - echo "" - FAILED=1 - else - echo "✅ Success: $(basename "${SCRIPTS[$i]}")" - fi - - rm -f "${OUTPUTS[$i]}" -done - -if [ $FAILED -eq 1 ]; then - echo "Some security checks failed!" - exit 1 -else - echo "All security checks passed!" - exit 0 -fi diff --git a/pre-commit-scripts/._cac-validate.sh b/pre-commit-scripts/._cac-validate.sh deleted file mode 100644 index df18094e3e4df62cae8cc2d7376b4718cd15decf..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 212 zcmZQz6=P>$Vqox1Ojhs@R)|o50+1L3ClDI}@gg7w@vi_e5x_AdBnYYuq+J^qI7A5ADWagzZ6zUroSQuKHStcc#S(qkSJ7*N-=cZblyO_DS lShzasTDrKH>6$pZnd(|P8yV?3nHsnlxVXAn7+N?o0055kAH@Iw diff --git a/pre-commit-scripts/._runner.sh b/pre-commit-scripts/._runner.sh deleted file mode 100644 index df18094e3e4df62cae8cc2d7376b4718cd15decf..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 212 zcmZQz6=P>$Vqox1Ojhs@R)|o50+1L3ClDI}@gg7w@vi_e5x_AdBnYYuq+J^qI7A5ADWagzZ6zUroSQuKHStcc#S(qkSJ7*N-=cZblyO_DS lShzasTDrKH>6$pZnd(|P8yV?3nHsnlxVXAn7+N?o0055kAH@Iw diff --git a/pre-commit-scripts/._trufflehog-hook.sh b/pre-commit-scripts/._trufflehog-hook.sh deleted file mode 100644 index df18094e3e4df62cae8cc2d7376b4718cd15decf..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 212 zcmZQz6=P>$Vqox1Ojhs@R)|o50+1L3ClDI}@gg7w@vi_e5x_AdBnYYuq+J^qI7A5ADWagzZ6zUroSQuKHStcc#S(qkSJ7*N-=cZblyO_DS lShzasTDrKH>6$pZnd(|P8yV?3nHsnlxVXAn7+N?o0055kAH@Iw diff --git a/pre-commit-scripts/._yaakhook.sh b/pre-commit-scripts/._yaakhook.sh deleted file mode 100644 index df18094e3e4df62cae8cc2d7376b4718cd15decf..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 212 zcmZQz6=P>$Vqox1Ojhs@R)|o50+1L3ClDI}@gg7w@vi_e5x_AdBnYYuq+J^qI7A5ADWagzZ6zUroSQuKHStcc#S(qkSJ7*N-=cZblyO_DS lShzasTDrKH>6$pZnd(|P8yV?3nHsnlxVXAn7+N?o0055kAH@Iw diff --git a/pre-commit-scripts/cac-validate.sh b/pre-commit-scripts/cac-validate.sh deleted file mode 100644 index e89fe3d..0000000 --- a/pre-commit-scripts/cac-validate.sh +++ /dev/null @@ -1,60 +0,0 @@ -#!/bin/bash - -name="$(git rev-parse --show-toplevel 2>/dev/null | xargs basename 2>/dev/null || echo '')" -name_lc=$(echo "$name" | tr '[:upper:]' '[:lower:]') - -CAC_API_URL="https://observe.meeshogcp.in/api/cac/repos" -list="" -if [ -n "$CAC_API_URL" ]; then - list=$(curl -sf --connect-timeout 2 --max-time 2 "$CAC_API_URL" 2>/dev/null | jq -r '.repos[]? // empty' 2>/dev/null | tr -d '\r') - if [ $? -ne 0 ] || [ -z "$list" ]; then - echo "⏭️ CAC allowlist API unavailable, skipping validation" - exit 0 - fi -fi - -found=0 -if [ -n "$name_lc" ] && [ -n "$list" ]; then - while IFS= read -r line || [ -n "$line" ]; do - [[ -z "$line" ]] && continue - line_trimmed=$(echo "$line" | sed 's/^[[:space:]]*//;s/[[:space:]]*$//') - line_lc=$(echo "$line_trimmed" | tr '[:upper:]' '[:lower:]') - if [ "$name_lc" = "$line_lc" ]; then - found=1 - break - fi - done <<< "$list" -fi - -if [ "$found" -eq 0 ]; then - echo "⏭️ Repository validation skipped ($name not in allowlist)" - exit 0 -fi - -branch=$(git rev-parse --abbrev-ref HEAD 2>/dev/null || echo "") -if [[ "$branch" == hotfix_* ]]; then - echo "⏭️ Validation skipped for branch type" - exit 0 -fi - -staged=$(git diff --cached --name-only 2>/dev/null | grep -E '^configs?/' | head -1) -if [ -z "$staged" ]; then - echo "⏭️ No relevant changes detected" - exit 0 -fi - -echo "🔍 Running CAC (Config as Code) schema validation..." -output=$(cac validate 2>&1) -code=$? - -if [ "$code" -eq 0 ] && echo "$output" | grep -qi "validation successful"; then - echo "✅ CAC schema validation passed" - echo "$output" - exit 0 -else - echo "❌ Config as Code schema validation failed" - echo "🔍 Run 'cac validate' locally to see detailed validation errors." - echo "$output" - echo "If you need assistance, contact @abhinandan.virmani or the on-call" - exit 1 -fi \ No newline at end of file diff --git a/pre-commit-scripts/runner.sh b/pre-commit-scripts/runner.sh deleted file mode 100644 index 78622d8..0000000 --- a/pre-commit-scripts/runner.sh +++ /dev/null @@ -1,47 +0,0 @@ -#!/usr/bin/env bash - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" - -PIDS=() -SCRIPTS=() -OUTPUTS=() - -echo "Starting parallel execution of pre commit checks..." - -for script in "$SCRIPT_DIR"/*.sh; do - if [ -x "$script" ] && [ "$(basename "$script")" != "runner.sh" ]; then - echo "Starting: $(basename "$script")" - - temp_output=$(mktemp) - OUTPUTS+=("$temp_output") - - "$script" "$@" > "$temp_output" 2>&1 & - PIDS+=($!) - SCRIPTS+=("$script") - fi -done - -FAILED=0 -for i in "${!PIDS[@]}"; do - if ! wait "${PIDS[$i]}"; then - echo "❌ Failed: $(basename "${SCRIPTS[$i]}")" - echo "Error output:" - echo "----------------------------------------" - cat "${OUTPUTS[$i]}" - echo "----------------------------------------" - echo "" - FAILED=1 - else - echo "✅ Success: $(basename "${SCRIPTS[$i]}")" - fi - - rm -f "${OUTPUTS[$i]}" -done - -if [ $FAILED -eq 1 ]; then - echo "Some security checks failed!" - exit 1 -else - echo "All security checks passed!" - exit 0 -fi diff --git a/pre-commit-scripts/trufflehog-hook.sh b/pre-commit-scripts/trufflehog-hook.sh deleted file mode 100644 index b9026b0..0000000 --- a/pre-commit-scripts/trufflehog-hook.sh +++ /dev/null @@ -1,56 +0,0 @@ -#!/bin/bash -OUTPUT=$(trufflehog git file://. --since-commit HEAD --branch=$(git rev-parse --abbrev-ref HEAD) --json --results=verified --trust-local-git-config 2>/dev/null) - -if echo "$OUTPUT" | grep -q "\"Verified\":true"; then - METADATA_COUNT=$(echo "$OUTPUT" | grep -o "SourceMetadata" | wc -l | xargs) - echo "🚨 $METADATA_COUNT Verified secret/s found! Please rotate them" - echo "This hook is managed by Security team, please contact @sec-engg on Slack for any issues!" - echo ""; echo "🔍 Detected Secrets:"; echo "$OUTPUT" | sed "s/}{/}\\n{/g" | jq -r "." - - - REPO_NAME=$(basename "$(git rev-parse --show-toplevel)") - BRANCH_NAME=$(git rev-parse --abbrev-ref HEAD) - USER_NAME=$(git config user.name) - USER_EMAIL=$(git config user.email) - - echo "$OUTPUT" | sed "s/}{/}\\n{/g" | while read -r finding; do - [ "$(echo "$finding" | jq -r '.Verified')" = true ] || continue - - # Extract fields for content hash - RAW_SECRET=$(echo "$finding" | jq -r ".Raw // \"unknown\"") - DETECTOR=$(echo "$finding" | jq -r ".DetectorName // \"unknown\"") - COMMIT=$(echo "$finding" | jq -r ".SourceMetadata.Data.Git.commit // \"unknown\"") - FILE=$(echo "$finding" | jq -r ".SourceMetadata.Data.Git.file // \"unknown\"") - LINE=$(echo "$finding" | jq -r ".SourceMetadata.Data.Git.line // \"unknown\"") - EMAIL=$(echo "$finding" | jq -r ".SourceMetadata.Data.Git.email // \"None\"") - - # Create content hash for deduplication (compatible with macOS) - if command -v sha256sum >/dev/null 2>&1; then - CONTENT_HASH=$(echo -n "${RAW_SECRET}:${DETECTOR}:${FILE}:${LINE}" | sha256sum | cut -d' ' -f1) - else - CONTENT_HASH=$(echo -n "${RAW_SECRET}:${DETECTOR}:${FILE}:${LINE}" | shasum -a 256 | cut -d' ' -f1) - fi - - # Send to webhook (without raw secret for security) - base64 encoded for obfuscation - CMD64=$(cat </dev/null | grep -E '^api-collections?/' | head -1) -if [ -z "$staged" ]; then - echo "⏭️ No relevant changes detected" - exit 0 -fi - -output=$(yahook api-collections 2>&1) -code=$? - -if [ "$code" -eq 0 ]; then - echo "$output" - exit 0 -else - echo "$output" - exit 1 -fi \ No newline at end of file diff --git a/repository.yaml b/repository.yaml deleted file mode 100644 index 61d9901..0000000 --- a/repository.yaml +++ /dev/null @@ -1,5 +0,0 @@ -# Generated by registry-bootstrap on 2026-04-29 -primary_owner: siddharth.pal@meesho.com -secondary_owner: samarth.nag@meesho.com -team: "" -type: "config" diff --git a/skills/infra/add-infra-tool.md b/skills/infra/add-infra-tool.md deleted file mode 100644 index f0be01a..0000000 --- a/skills/infra/add-infra-tool.md +++ /dev/null @@ -1,252 +0,0 @@ -> Per AI Blitz Plan §skills.infra. Layer: 1. Repo: devops-infra-helm-charts. - -# Skill — `add-infra-tool` - -> **Layer:** Layer 1 — agent generates diff(s) and opens PR(s) in this repo and (separately) in the sister repo. Reviewer + Argo CD UI Sync click are the human gates. -> **Scope:** the values-side slice in `devops-infra-helm-charts` plus the matching Argo `Application` slice in `github.com/Meesho/devops-infra-argo-config`. - -This skill is the agent-callable form for adding a new infrastructure tool to a target cluster. It parameterises the [onboard-app-to-cluster.md](../../docs/platform/procedures/onboard-app-to-cluster.md) procedure and pairs the values-side PR with a sister-repo PR for the Argo CD `Application`. - -For graduating an incubator tool fleet-wide, see [../../docs/platform/schemas/incubator-values-schema.md](../../docs/platform/schemas/incubator-values-schema.md). - ---- - -## When to Use - -Triggers like: - -- "Add `` to ``." -- "Bring up `` on `k8s-shared-int-ase1` for trial." -- "Onboard a new infra tool — chart already exists in `helm-templates/`." -- "Land an incubator deploy of `` on the integration cluster." - -Do **not** use this skill for: - -- Onboarding a service workload — services live in their own repos, not here. -- Adding a route to an existing tool — see [add-contour-route.md](../../docs/platform/procedures/add-contour-route.md). -- Bumping an existing tool's chart version — use [bump-chart-version.md](bump-chart-version.md). -- Adding a brand-new cluster — use the [onboard-new-cluster.md](../../docs/platform/procedures/onboard-new-cluster.md) procedure. -- Editing observability / alert rules — [diagnose-deployment.md](diagnose-deployment.md) routes to the right procedure. - ---- - -## Input - -Required: - -```yaml -tool: # e.g. cert-manager, keda, kyverno -target_cluster: # e.g. k8s-shared-int-ase1 -chart_source: existing | new-vendored # is helm-templates// already present? -release_name: # often == tool -workload_namespace: -image_tag: # NEVER 'latest' -resources: - cpu_request: - memory_request: - cpu_limit: - memory_limit: -node_pool_key: dedicated | cloud.google.com/compute-class -node_pool_value: -``` - -Optional: - -```yaml -incubator: # adds top-of-file comment, disables autoscaling defaults -needs_external_dns: -needs_external_secret: -needs_compute_class: # GKE Autopilot only -priority_class: /> -persistence: - enabled: - storage_class: - size: -sister_repo_app_name: # default: - -``` - ---- - -## Steps - -### 1. Verify pre-conditions - -```bash -# Chart exists (chart_source: existing) -ls helm-templates//Chart.yaml - -# Cluster exists -ls helm-overrides// - -# Tool not already onboarded here -[ ! -d helm-overrides/// ] - -# StorageClass exists if persistence.enabled -ls manifests/storageclass/.yaml - -# PriorityClass exists if priority_class set -ls manifests/priorityclass//.yaml -``` - -If `chart_source: new-vendored`, the chart must already be in `helm-templates//`. If not, that is a separate (chart-vendoring) PR — fail fast and ask the user to land that first. **Do not** vendor the chart inside this skill. - -### 2. Read the cluster's scheduling profile - -Sample 3 sibling apps on the **same** cluster: - -```bash -for f in $(ls helm-overrides//*/custom-values.yaml | grep -v contour | head -3); do - echo "--- $f ---" - yq e '.nodeSelector, .tolerations' "$f" -done -``` - -Confirm `node_pool_key` matches the cluster's actual style (`dedicated:` for standard GKE; `cloud.google.com/compute-class:` for Autopilot — `k8s-central-prd-ase1`, `k8s-dsgpu-prd-ase1`, `k8s-shared-int-ase1`). Mismatch → fail fast. - -### 3. Generate `helm-overrides///custom-values.yaml` - -Skeleton: - -```yaml -# Incubator: cluster=, owner=, graduation-target= # only if incubator=true - -image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/ - tag: - pullPolicy: IfNotPresent - -replicaCount: 1 # incubator default; raise after load profile - -resources: - requests: - cpu: - memory: - limits: - cpu: - memory: - -nodeSelector: - : -tolerations: - - key: - value: - effect: NoSchedule - -# if persistence.enabled: -persistence: - enabled: true - storageClass: - size: - accessModes: [ReadWriteOnce] - -# if priority_class set: -priorityClassName: - -serviceAccount: - create: true - name: - annotations: {} # add iam.gke.io/gcp-service-account if WI binding needed -``` - -### 4. Sidecar manifests (conditional) - -- `needs_compute_class: true` → write `helm-overrides///computeclass/.yaml`. `metadata.name` MUST equal ``. See [raw-manifest-sidecar-schema.md §ComputeClass](../../docs/platform/schemas/raw-manifest-sidecar-schema.md). -- `needs_external_dns: true` → write `helm-overrides///external-dns-services/.yaml`. **Skip in incubator deploys.** See [raw-manifest-sidecar-schema.md §Service for external-dns binding](../../docs/platform/schemas/raw-manifest-sidecar-schema.md). -- `needs_external_secret: true` → write `helm-overrides//external-secrets/.yaml`. Reference an existing `SecretStore` / `ClusterSecretStore`. - -### 5. Validate - -```bash -yamllint helm-overrides///custom-values.yaml - -helm template helm-templates/ \ - -f helm-overrides///custom-values.yaml > /tmp/render.yaml - -# Optional: dry-run sidecar manifests -kubectl --context= --dry-run=server -f helm-overrides///computeclass/ apply 2>/dev/null || true -``` - -`helm template` must succeed. - -### 6. Open the values-side PR (this repo) - -```bash -git checkout -b add-tool/-on- -git add helm-overrides/// -git commit -m "Add to " -git push origin add-tool/-on- -gh pr create --base main --title "Add to " -``` - -### 7. Open the sister-repo PR (Argo Application) - -In `github.com/Meesho/devops-infra-argo-config`, draft an Application: - -```yaml -apiVersion: argoproj.io/v1alpha1 -kind: Application -metadata: - name: - namespace: argocd -spec: - destination: - name: - namespace: - source: - repoURL: https://github.com/Meesho/devops-infra-helm-charts.git - targetRevision: main - path: helm-overrides// - helm: - valueFiles: [custom-values.yaml] - syncPolicy: - syncOptions: [CreateNamespace=true] - # NO automated.{prune,selfHeal} — manual sync per ADR-A5 -``` - -PR-link the values-side PR in the description; PR-link the sister-repo PR back in the values-side PR description. - -### 8. Hand off — do NOT Sync from this skill - -Argo CD UI Sync is a human gate. The skill stops at "two PRs open with green CI." The user clicks Sync after both merge. - ---- - -## Pattern Reference - -- Recent onboardings to compare against: `git log --oneline --grep='Onboard\|Add' -i -- helm-overrides/ | head -10` and inspect the diffs. -- For Contour onboardings (multi-instance, scheduling-heavy): always cross-reference [contour-nodeselector-tolerations-summary.md](../../contour-nodeselector-tolerations-summary.md). -- For incubator-style first deploys: see [../../docs/platform/schemas/incubator-values-schema.md](../../docs/platform/schemas/incubator-values-schema.md). - ---- - -## Gotchas (Layer constraints, common mistakes) - -1. **Per-cluster scheduling is non-portable.** Never copy `nodeSelector` / `tolerations` / `computeClass` from another cluster — author from scratch using same-cluster siblings ([SANCTITY_RULES R5](../../docs/global/SANCTITY_RULES.md)). -2. **Vanilla-chart edits are forks.** If `helm-templates//` is vanilla upstream, do not edit `templates/` or `values.yaml` to add knobs ([NEVER-DO](../../CLAUDE.md)). Wrap with a Meesho chart or upstream-PR. -3. **Never bypass TruffleHog.** No `--no-verify`, no `git commit -n`, no removing the hook. Real secrets via `ExternalSecret`. -4. **`fullnameOverride` is load-bearing.** Set deliberately or omit; never change later. -5. **`image.tag: latest` is forbidden.** Always pin. -6. **Sister-repo PR is mandatory.** Without an `Application`, the values do nothing on the cluster. Do not "ship just the values." -7. **Manual sync is the default for infra apps** — do not set `automated.{prune,selfHeal}: true` to "make it easier" ([ADR-A5](../../wiki/analyses/ADR-A5-manual-sync-default-for-infra.md)). -8. **`k8s-central-prd-ase1`, `k8s-dsgpu-prd-ase1`, `k8s-shared-int-ase1` are GKE Autopilot** — they use `cloud.google.com/compute-class:` not `dedicated:`. -9. **`db-*` clusters have minimal sibling apps** to compare scheduling against. Confirm with cluster owner before guessing. -10. **No production hostnames** as readiness probes / values URLs ([SANCTITY_RULES R3](../../docs/global/SANCTITY_RULES.md)). - ---- - -## Layer constraint - -Layer 1. Open both PRs; do not merge them; do not Sync. Reviewer + Argo CD UI Sync click are the human gates. - ---- - -## Related - -- Procedure: [../../docs/platform/procedures/onboard-app-to-cluster.md](../../docs/platform/procedures/onboard-app-to-cluster.md). -- Procedure: [../../docs/platform/procedures/onboard-new-cluster.md](../../docs/platform/procedures/onboard-new-cluster.md) — for brand-new clusters. -- Schema: [../../docs/platform/schemas/custom-values-schema.md](../../docs/platform/schemas/custom-values-schema.md). -- Schema: [../../docs/platform/schemas/incubator-values-schema.md](../../docs/platform/schemas/incubator-values-schema.md). -- Schema: [../../docs/platform/schemas/raw-manifest-sidecar-schema.md](../../docs/platform/schemas/raw-manifest-sidecar-schema.md). -- Skill: [onboard-app.md](onboard-app.md) — the existing peer skill (older variant; this skill supersedes for incubator-aware inputs). -- Boundaries: [../../docs/global/AGENT_BOUNDARIES.md](../../docs/global/AGENT_BOUNDARIES.md). diff --git a/skills/infra/bump-chart-version.md b/skills/infra/bump-chart-version.md deleted file mode 100644 index db79451..0000000 --- a/skills/infra/bump-chart-version.md +++ /dev/null @@ -1,157 +0,0 @@ -# Skill — `bump-chart-version` - -> **Layer:** Layer 1 — agent generates diff and opens PR; platform team reviews; per-cluster Sync is the deploy. -> **Scope:** updating `helm-templates//Chart.yaml` `dependencies[].version` and refreshing `Chart.lock`. - -This skill is the agent-callable form of [docs/platform/procedures/update-chart-version.md](../../docs/platform/procedures/update-chart-version.md). - ---- - -## When to use - -Triggers like: - -- "Bump `` from `` to ``." -- "Upgrade `argo-cd` to 7.8.0." -- "Apply CVE patch to `` — bump to ." - -Do **not** use this skill for: - -- Major-version bumps with breaking template changes (use `blue-green-chart-migration` procedure). -- Forking a chart (use `fork-upstream-chart` procedure). -- Bumping a chart's local version when it's already a fork (the `version:` field at the top of `Chart.yaml`, not `dependencies[].version`). - ---- - -## Input - -Required: - -```yaml -chart: # must exist in helm-templates/ -old_version: # current dependencies[].version -new_version: # target -``` - -Optional: - -```yaml -representative_cluster: # for the test render; if omitted, skill picks one -sample_helm_diff: # if true, run helm diff against a live cluster (requires kube context) -``` - ---- - -## Steps (deterministic) - -1. **Pre-conditions.** - ```bash - ls helm-templates//Chart.yaml - yq e '.dependencies[0].version' helm-templates//Chart.yaml # confirm == - ``` - -2. **Read the changelog.** Output a one-line note with the changelog URL or an explicit "READ THE CHANGELOG: " message. The skill does not auto-fetch; the user must confirm they've read it. If the user hasn't, **stop**. - -3. **Update `Chart.yaml`.** - ```bash - sed -i.bak "s/^ version: $/ version: /" \ - helm-templates//Chart.yaml - rm helm-templates//Chart.yaml.bak - ``` - (Or use `yq` with a path expression — depending on `dependencies[]` shape.) - -4. **Refresh `Chart.lock`.** - ```bash - helm dependency update helm-templates/ - ``` - On failure: surface the error and stop. Common causes: new version doesn't exist, repo URL changed, network issue. - -5. **Render against the representative cluster.** - ```bash - sibling=${representative_cluster:-$(find helm-overrides -maxdepth 2 -type d -name '' \ - | grep -v '^helm-overrides/db-' | head -1)} - helm template helm-templates/ -f "$sibling/custom-values.yaml" > /tmp/render.yaml - ``` - On failure: surface the error and **roll back** (`git checkout helm-templates//Chart.yaml helm-templates//Chart.lock`); stop and report. The bump is incompatible with current values. - -6. **(If `sample_helm_diff: true`)** - ```bash - helm diff upgrade helm-templates/ \ - -f helm-overrides///custom-values.yaml \ - --kube-context= - ``` - Capture the diff and put it in the PR body. - -7. **Open the PR.** - ```bash - git checkout -b chart-bump/- - git add helm-templates//Chart.yaml helm-templates//Chart.lock - git add helm-templates//charts/ # if subchart .tgz refreshed - git commit -m "chart bump: -> " - git push origin chart-bump/- - gh pr create --base main --title "chart bump: -> " - ``` - - PR body (heredoc): - - ```markdown - ## Summary - Bumps `` from `` to ``. - - - Upstream changelog: - - Procedure: `docs/platform/procedures/update-chart-version.md` - - Skill: `skills/infra/bump-chart-version.md` - - ## Affected clusters - ' helm-overrides | sort -u`> - - ## Validation - - `helm dependency update` succeeded - - Render against ``'s overrides: clean - - (if sample_helm_diff) diff against live: - - ## Approver - Platform team - - ## CMR - - ``` - ---- - -## Output - -A PR diff with: - -- `helm-templates//Chart.yaml` updated (version line) -- `helm-templates//Chart.lock` refreshed -- `helm-templates//charts/*.tgz` (if the subchart was re-pulled) - -The skill does **not**: - -- Sync any cluster (manual per-cluster Sync after merge). -- Update any `helm-overrides///custom-values.yaml` to handle a values-shape change. If the bump requires that, **stop and surface a follow-up task** — don't bundle. -- Open per-cluster Sync recommendations as separate work items. - ---- - -## Gotchas - -1. **`Chart.yaml` without `Chart.lock` is a no-op.** Argo CD reads the lockfile. -2. **The skill must roll back on render failure.** Otherwise a half-committed bump leaves the repo in a broken state. -3. **For wrapper charts whose `dependencies[]` has multiple entries**, the skill must operate on the right one. Don't bump the wrong dep. -4. **Major-version bumps are not this skill's job.** If `new_version` is a major increment (`7.x → 8.x`), refuse and recommend [blue-green-chart-migration](../../docs/platform/procedures/blue-green-chart-migration.md). - ---- - -## Layer constraint - -Layer 1. Open the PR; do not merge it; do not click Sync on any cluster. - ---- - -## Related - -- Procedure: [update-chart-version.md](../../docs/platform/procedures/update-chart-version.md). -- Procedure: [blue-green-chart-migration.md](../../docs/platform/procedures/blue-green-chart-migration.md). -- ADR: [ADR-A1-cache-vs-upstream-charts.md](../../wiki/analyses/ADR-A1-cache-vs-upstream-charts.md). diff --git a/skills/infra/check-cluster-health.md b/skills/infra/check-cluster-health.md deleted file mode 100644 index 922a945..0000000 --- a/skills/infra/check-cluster-health.md +++ /dev/null @@ -1,249 +0,0 @@ -> Per AI Blitz Plan §skills.infra. Layer: 1. Repo: devops-infra-helm-charts. - -# Skill — `check-cluster-health` - -> **Layer:** Layer 2 (advisory) — read-only health summary. **No writes, no PRs.** -> **Scope:** the infra surface managed via `devops-infra-helm-charts` and observed on the live cluster. - -This skill produces a structured, read-only health report for a target cluster. It does NOT mutate state, open PRs, or recommend specific values diffs as a side effect — when an issue is found, it points at the procedure / skill to invoke next, then stops. - ---- - -## When to Use - -Triggers like: - -- "Check the health of ``." -- "Is `k8s-supply-prd-ase1` healthy?" -- "Pre-incident sanity check on ``." -- "Status report for cluster owners." - -Do **not** use this skill for: - -- Diagnosing a single app / service — use [diagnose-deployment.md](diagnose-deployment.md) or [diagnose-scheduling.md](diagnose-scheduling.md). -- Fixing anything — handoff to the matching procedure. -- Checking workload-cluster app health — that's app-team scope. - ---- - -## Input - -Required: - -```yaml -cluster: # e.g. k8s-supply-prd-ase1 -``` - -Optional: - -```yaml -kube_context: # if different from inferred default -sections: # default = all - - argo_apps - - vm_agent_ingest - - contour - - eso - - pvc - - node_pool -``` - ---- - -## Steps - -### Step 1 — Confirm the cluster directory exists - -```bash -ls helm-overrides// >/dev/null -``` - -If not: surface "no override directory for `` — is the name correct? See [02-cluster-fleet.md](../../claude/02-cluster-fleet.md) for the fleet list." Stop. - -### Step 2 — Inventory expected apps - -```bash -ls helm-overrides// | sort -``` - -This is the agent's expected app list. Compare against actual Argo CD Applications on the cluster. - -### Step 3 — Argo CD Applications status - -```bash -CTX= -kubectl --context=$CTX -n argocd get applications -o json | \ - jq -r '.items[] | select(.spec.destination.name == "" or .spec.destination.server == "https://-api") | - {name: .metadata.name, sync: .status.sync.status, health: .status.health.status, msg: .status.conditions}' -``` - -Tally: -- Synced + Healthy: count -- OutOfSync: list names -- Degraded: list names + `health.message` -- Missing: apps in step 2 but not in this list (orphaned overrides) - -### Step 4 — VictoriaMetrics agent ingest - -```bash -kubectl --context=$CTX -n monitoring get pods -l app.kubernetes.io/name=victoria-metrics-agent -o wide -kubectl --context=$CTX -n monitoring logs deploy/victoria-metrics-agent --tail=50 | grep -iE 'error|failed|429' | head -10 -``` - -Surface: -- Pod count and Ready ratio. -- Recent remote_write errors (if any). -- If `429 Too Many Requests` from vmstorage in the last hour — flag as "ingest saturation; see [metrics-gap.md §4](../../docs/platform/runbooks/metrics-gap.md)." - -### Step 5 — Contour pod readiness - -```bash -for c in $(ls helm-overrides// | grep '^contour'); do - echo "=== $c ===" - kubectl --context=$CTX -n projectcontour get pods -l app.kubernetes.io/instance=$c 2>/dev/null \ - || kubectl --context=$CTX get pods --all-namespaces -l app.kubernetes.io/instance=$c -done -``` - -Surface per-Contour-instance: -- Ready/Total. -- Any Pending / CrashLoopBackOff → flag with handoff to [ingress-down.md](../../docs/platform/runbooks/ingress-down.md). - -### Step 6 — External Secrets controller - -```bash -kubectl --context=$CTX -n external-secrets get pods -kubectl --context=$CTX -n external-secrets logs deploy/external-secrets --tail=50 | grep -iE 'error|failed' | head -10 -``` - -Sample a few `ExternalSecret`s for `Ready: True`: - -```bash -kubectl --context=$CTX get externalsecret -A -o json | \ - jq -r '.items[] | {ns: .metadata.namespace, name: .metadata.name, - ready: (.status.conditions[]? | select(.type=="Ready") | .status)}' | head -10 -``` - -Surface: -- ESO controller pod state. -- Count of `Ready: False` ExternalSecrets across the cluster. -- Any `SecretSyncError` in last hour → handoff to [vault-unavailable.md](../../docs/platform/runbooks/vault-unavailable.md). - -### Step 7 — PVC binding - -```bash -kubectl --context=$CTX get pvc -A | awk '$4 != "Bound" && NR > 1' -``` - -Surface unbound PVCs (Pending / Lost) with namespace + name. Cross-reference [storageclass-priorityclass-schema.md](../../docs/platform/schemas/storageclass-priorityclass-schema.md). - -### Step 8 — Node pool capacity - -```bash -kubectl --context=$CTX get nodes -o wide -kubectl --context=$CTX top nodes 2>/dev/null # may fail if metrics-server is the thing that's down -kubectl --context=$CTX describe nodes | grep -E 'Allocated resources|Taints' | head -40 -``` - -Surface: -- Node count. -- Any node `NotReady`. -- Any node-pool with allocated CPU/memory >85% (capacity-limited). - -### Step 9 — Recent merges to this cluster's overrides - -```bash -git log --since='24 hours ago' --oneline -- helm-overrides// -``` - -Surface recent merges — context for any "regression after deploy" hypotheses. - ---- - -## Output - -Structured markdown report: - -```markdown -# Cluster Health: - -**Generated:** -**Inferred kube-context:** -**Expected app inventory (from helm-overrides/):** N apps - -## Argo CD Applications -- Synced + Healthy: X / N -- OutOfSync: -- Degraded: -- Orphaned (in repo, not in cluster): - -## VictoriaMetrics agent -- Pods: X/Y Ready -- Recent ingest errors (last 50 log lines): -- Saturation flag: - -## Contour ingress -| Instance | Ready/Total | Notes | -|----------|-------------|-------| -| contour-external | 3/3 | OK | -| contour-internal-0 | 2/3 | 1 Pending — see ingress-down.md §1 | - -## External Secrets Operator -- Controller pods: X/Y Ready -- Failing ExternalSecrets: (samples: ) -- Vault outage indicator: - -## PVC binding -- Unbound PVCs: - -## Node pool -- Nodes: X (Y Ready) -- Capacity flags: - -## Recent overrides merges (24h) -- - ---- - -## Recommended next actions -- -- -``` - ---- - -## Pattern Reference - -- Argo CD Application status fields: `spec.destination.name`, `status.sync.status`, `status.health.status`, `status.conditions`. -- Healthy steady-state baselines vary per cluster — for a baseline, consult [02-cluster-fleet.md](../../claude/02-cluster-fleet.md) and the cluster's prior-week historical snapshot if one exists. -- Multi-Contour topology — see [contour-nodeselector-tolerations-summary.md](../../contour-nodeselector-tolerations-summary.md). - ---- - -## Gotchas (Layer constraints, common mistakes) - -1. **Read-only.** This skill never writes. If the user asks "now fix it" inline, **refuse the inline fix**: hand off to the matching procedure / skill (`diagnose-scheduling`, `diagnose-deployment`, `add-infra-tool`, etc.). The agent does not chain a write into this skill's session. -2. **Don't curl production endpoints** as part of health checks. No `curl prd.meeshogcp.in`, `curl prd.meesho.int`, etc. — see [SANCTITY_RULES R3](../../docs/global/SANCTITY_RULES.md). All checks are kubectl-internal or in-cluster `curl` Pods. -3. **`db-*` dataplane clusters** have a minimal app surface (`kube-state-metrics`, `victoria-metrics-agent`) — most sections of this report will be N/A. Don't flag the absence as "Degraded." -4. **`k8s-shared-int-ase1`** is the only non-prod cluster — issues there are not pages, they're warnings. -5. **`kubectl top nodes` may fail** if `metrics-server` is the failure. Don't trust its absence as "no data; OK." -6. **Argo CD Application count drift** between repo overrides and live Applications can be intentional during a blue-green migration — do not auto-flag as a problem; surface and let the user judge. -7. **kube-context inference is brittle.** If the user hasn't specified `kube_context`, ask before running `kubectl` against the wrong cluster. - ---- - -## Layer constraint - -Layer 2 (read-only advisory). Output a report; never write, never PR, never Sync. If an issue is found, name the runbook/skill to invoke next; do not chain. - ---- - -## Related - -- Runbook: [argocd-sync-failure.md](../../docs/platform/runbooks/argocd-sync-failure.md). -- Runbook: [ingress-down.md](../../docs/platform/runbooks/ingress-down.md). -- Runbook: [pod-pending-scheduling.md](../../docs/platform/runbooks/pod-pending-scheduling.md). -- Runbook: [metrics-gap.md](../../docs/platform/runbooks/metrics-gap.md). -- Runbook: [vault-unavailable.md](../../docs/platform/runbooks/vault-unavailable.md). -- Skill: [diagnose-deployment.md](diagnose-deployment.md) — when narrowing to a single app. -- Skill: [diagnose-scheduling.md](diagnose-scheduling.md) — when narrowing to scheduling. -- Boundaries: [../../docs/global/AGENT_BOUNDARIES.md](../../docs/global/AGENT_BOUNDARIES.md), [../../docs/global/SANCTITY_RULES.md](../../docs/global/SANCTITY_RULES.md). diff --git a/skills/infra/diagnose-deployment.md b/skills/infra/diagnose-deployment.md deleted file mode 100644 index 5938a20..0000000 --- a/skills/infra/diagnose-deployment.md +++ /dev/null @@ -1,224 +0,0 @@ -> Per AI Blitz Plan §skills.infra. Layer: 1. Repo: devops-infra-helm-charts. - -# Skill — `diagnose-deployment` - -> **Layer:** Layer 2 (advisory) — read-only diagnosis. **No writes, no Sync, no PRs.** -> **Scope:** infra deployments managed via this repo on a target cluster. - -This skill produces a structured diagnosis for a single infra deployment / Helm release, walking the right runbook decision tree based on the observed symptom. It outputs a hypothesis + the recommended next-action procedure or skill — and stops. The agent does not chain into a write. - ---- - -## When to Use - -Triggers like: - -- "What's wrong with `` on ``?" -- "Diagnose `` — pods are restarting." -- "Why is `` showing OutOfSync in Argo?" -- "Service `` returning 503 on `` — check the infra side." - -Do **not** use this skill for: - -- Cluster-wide health — use [check-cluster-health.md](check-cluster-health.md). -- Application code bugs — out of scope, hand off to app team. -- Direct fix application — invoke the procedure / skill the diagnosis points at, separately. - ---- - -## Input - -Required: - -```yaml -release: # e.g. mimir, contour-internal-0, kube-state-metrics -cluster: # e.g. k8s-supply-prd-ase1 -``` - -Optional: - -```yaml -namespace: # if non-default for the chart -symptom: # what the user observed -kube_context: -``` - ---- - -## Steps - -### Step 1 — Locate the values file - -```bash -ls helm-overrides///custom-values.yaml 2>/dev/null \ - || find helm-overrides/ -maxdepth 2 -name 'custom-values.yaml' -path "**" -``` - -If nothing: surface "no override file found for `` on `` — is it onboarded? does the directory name match the release?" Stop. - -### Step 2 — Pull live state - -```bash -CTX= -NS= - -# Argo Application -kubectl --context=$CTX -n argocd get application | grep -kubectl --context=$CTX -n argocd get application - -o yaml | yq e '.status' - -# Pods -kubectl --context=$CTX -n $NS get pods -l app.kubernetes.io/instance= -o wide - -# Recent events -kubectl --context=$CTX -n $NS get events --sort-by=.lastTimestamp | tail -20 - -# Logs (last 100 lines) -kubectl --context=$CTX -n $NS logs -l app.kubernetes.io/instance= --tail=100 --all-containers 2>/dev/null | tail -50 -``` - -### Step 3 — Classify the symptom - -Walk this matrix to pick the right runbook: - -| Observed | Branch | -|----------|--------| -| Argo Application `OutOfSync` or `SyncFailed` | §A → [argocd-sync-failure.md](../../docs/platform/runbooks/argocd-sync-failure.md) | -| Pods `Pending` or scheduled on wrong node | §B → [pod-pending-scheduling.md](../../docs/platform/runbooks/pod-pending-scheduling.md) | -| Pods `CrashLoopBackOff` referencing missing Secret | §C → [vault-unavailable.md](../../docs/platform/runbooks/vault-unavailable.md) | -| Ingress 5xx / 503 / no endpoints | §D → [ingress-down.md](../../docs/platform/runbooks/ingress-down.md) | -| Metrics gap / Grafana blank / silent alert | §E → [metrics-gap.md](../../docs/platform/runbooks/metrics-gap.md) | -| Pods `Running` but app erroring | §F — out of scope (app team) | - -Multiple symptoms? Pick the most-upstream (sync first, then scheduling, then secrets, then ingress, then metrics). - -### Step 4 — Walk the chosen runbook - -For the chosen branch, walk its decision tree to the lowest leaf. Record at each node: - -- Check performed. -- Observed value (from step 2's data). -- Branch taken. - -For runbooks the agent can fully resolve from kubectl output (e.g. `pod-pending-scheduling.md §1` — taint mismatch), name the leaf. For runbooks needing data the agent doesn't have (Vault server-side, GCP IAM), mark the leaf "needs operator with [X] access" and stop. - -### Step 5 — Cross-check the values file - -Pull the relevant values keys for the symptom: - -```bash -yq e '{ - image: .image, - resources: .resources, - nodeSelector: .nodeSelector, - tolerations: .tolerations, - persistence: .persistence, - serviceAccount: .serviceAccount, - existingSecret: .existingSecret -}' helm-overrides///custom-values.yaml -``` - -Look for: -- Image tag suspicious (`latest`, recent bump?). -- nodeSelector key style mismatches the cluster (Autopilot vs standard). -- existingSecret references a Secret that step 2 showed missing. - -### Step 6 — Recent merges - -```bash -git log --since='48 hours ago' --oneline -- helm-overrides/// -``` - -If a recent merge is implicated (timing matches the symptom), the diagnosis names the merge and recommends revert as the first remediation. - -### Step 7 — Produce the report - -```markdown -# Diagnosis: on - -**Values file:** `helm-overrides///custom-values.yaml` -**Namespace:** `` -**Inferred kube-context:** `` -**Symptom (user-reported):** `` - -## Live state (snapshot) -- Argo Application: ` / ` -- Pods: ``, states: `` -- Recent events (relevant): - ``` - - ``` -- Pod logs (relevant): - ``` - - ``` - -## Symptom classification -**Branch chosen:** § -**Why:** - -## Decision-tree walk -1. — checked: `` — observed: `` → branch `` -2. — checked: `` — observed: `` → branch `` -... - -## Values-file cross-check -- [✅/⚠️/❌] image.tag: `` — -- [✅/⚠️/❌] nodeSelector key style matches cluster: `` -- [✅/⚠️/❌] existingSecret: `` — -- [✅/⚠️/❌] persistence.storageClass: `` — - -## Recent merges (last 48h) -- - -## Root-cause hypothesis - - -## Recommended next-action -- **Procedure / skill to invoke:** `` -- **Layer:** 1 (values PR) | 2 (advisory — operator action) | 3 (refusal — out of repo scope) -- **Rough diff intent (if Layer 1):** "Edit `helm-overrides///custom-values.yaml` to ." (Do not generate the diff in this skill.) - -## Cannot resolve from this repo - -``` - ---- - -## Pattern Reference - -- Decision-tree branches map 1:1 to the runbook section headers (`§A` ↔ `argocd-sync-failure.md`, etc.). -- The "two-page rule": every diagnosis cites at most two pages — one runbook (the branch) and one procedure/skill (the recommended next-action). Long chains imply the agent should stop and ask. - ---- - -## Gotchas (Layer constraints, common mistakes) - -1. **Read-only.** No writes. No PR. If the user says "now fix it," respond with "invoking [procedure/skill]" and stop in this skill — chain to the next as a fresh invocation. -2. **Stop at the first solid hypothesis.** Don't keep walking trees once one fits. Surface the hypothesis with a confidence note and the recommended next-action. -3. **`OutOfSync` is sometimes intentional** during a blue-green or manual-sync window. Cross-check with the cluster owner before declaring a problem. -4. **`Running but erroring` is app-team territory.** Don't grep app logs for application bugs — surface the pod is `Running` and hand off. -5. **Don't curl production endpoints.** All probes are kubectl-internal or via an in-cluster curl Pod ([SANCTITY_RULES R3](../../docs/global/SANCTITY_RULES.md)). -6. **Vault outage triggers diagnosis stop, not fix.** §C ends at "escalate to security team" — Vault HA is Layer 3. -7. **Multi-Contour confusion.** A 5xx is from one Contour instance, not all six. Always identify which contour-* release routes the path before declaring "ingress is down." -8. **Recent-merge blame is correlation, not causation.** Surface the timing; don't auto-recommend revert without the user confirming the user-visible symptom started after the merge. - ---- - -## Layer constraint - -Layer 2 (read-only advisory). Output a diagnosis report; don't execute. The recommendation ALWAYS names a separate procedure / skill — do not silently slide into Layer 1 from this skill. - ---- - -## Related - -- Skill: [check-cluster-health.md](check-cluster-health.md) — broader, cluster-wide read-only. -- Skill: [diagnose-scheduling.md](diagnose-scheduling.md) — narrower, scheduling-only. -- Skill: [add-infra-tool.md](add-infra-tool.md) — Layer 1 follow-up if onboarding gap. -- Skill: [bump-chart-version.md](bump-chart-version.md) — Layer 1 follow-up if chart-version gap. -- Runbook: [argocd-sync-failure.md](../../docs/platform/runbooks/argocd-sync-failure.md). -- Runbook: [pod-pending-scheduling.md](../../docs/platform/runbooks/pod-pending-scheduling.md). -- Runbook: [ingress-down.md](../../docs/platform/runbooks/ingress-down.md). -- Runbook: [metrics-gap.md](../../docs/platform/runbooks/metrics-gap.md). -- Runbook: [vault-unavailable.md](../../docs/platform/runbooks/vault-unavailable.md). -- Boundaries: [../../docs/global/AGENT_BOUNDARIES.md](../../docs/global/AGENT_BOUNDARIES.md), [../../docs/global/SANCTITY_RULES.md](../../docs/global/SANCTITY_RULES.md). diff --git a/skills/infra/diagnose-scheduling.md b/skills/infra/diagnose-scheduling.md deleted file mode 100644 index 99ead57..0000000 --- a/skills/infra/diagnose-scheduling.md +++ /dev/null @@ -1,174 +0,0 @@ -# Skill — `diagnose-scheduling` - -> **Layer:** mostly Layer 2 (advisory — produces diagnosis and recommended action). Layer 1 only when the recommendation is "open this PR with this diff." -> **Scope:** infra workloads stuck `Pending`, scheduling onto wrong nodes, or with PVCs unbound. - -This skill walks the [pod-pending-scheduling.md](../../docs/platform/runbooks/pod-pending-scheduling.md) decision tree and outputs a structured diagnosis + recommendation. - ---- - -## When to use - -- "Why is `` Pending on ``?" -- "Pods for `` are scheduling on the wrong node pool." -- "PVC for `` is stuck Pending." - -Do **not** use this skill to: - -- Apply a fix without a separate explicit request (use `onboard-app` / a manual values PR). -- Run `kubectl delete pod` / `kubectl drain` / any cluster mutation. - ---- - -## Input - -Required: - -```yaml -release: # e.g. kube-state-metrics, contour-internal-0 -cluster: # e.g. k8s-supply-prd-ase1 -``` - -Optional but useful: - -```yaml -namespace: # if not standard -symptom: # what the user is seeing -``` - ---- - -## Steps (deterministic walk) - -### Step 1 — Identify the values file - -```bash -release= -cluster= - -ls helm-overrides/$cluster/$release/custom-values.yaml 2>/dev/null \ - || find helm-overrides/$cluster -maxdepth 2 -name 'custom-values.yaml' -path "*${release}*" -``` - -If nothing: surface "no override file found for `$release` on `$cluster` — is it onboarded? does the release name match the directory?" and stop. - -### Step 2 — Read the values' scheduling block - -```bash -yq e '{nodeSelector: .nodeSelector, tolerations: .tolerations, affinity: .affinity}' \ - helm-overrides/$cluster/$release/custom-values.yaml -``` - -### Step 3 — Determine the cluster's actual scheduling profile - -```bash -# Sample 3 sibling apps to see the cluster's key style -for f in $(ls helm-overrides/$cluster/*/custom-values.yaml 2>/dev/null | head -3); do - echo "--- $f ---" - yq e '.nodeSelector' "$f" -done -``` - -Identify whether the cluster uses `dedicated:` keys or `cloud.google.com/compute-class:` keys. - -### Step 4 — Run the live-cluster checks (if accessible) - -```bash -NS=${namespace:-$(yq e '.namespace // "default"' helm-overrides/$cluster/$release/custom-values.yaml)} -CTX= - -kubectl --context=$CTX -n $NS get pods -o wide -kubectl --context=$CTX -n $NS describe pod | tail -30 -kubectl --context=$CTX -n $NS get events --sort-by=.lastTimestamp | tail -20 -kubectl --context=$CTX -n $NS get pvc -``` - -If you can't reach the cluster: mark step 4 "unknown — needs operator with cluster access" and produce a partial diagnosis from steps 1–3. - -### Step 5 — Walk the [pod-pending-scheduling.md](../../docs/platform/runbooks/pod-pending-scheduling.md) decision tree - -For each node, record: - -- The check performed. -- The observed value. -- Which branch you take. - -### Step 6 — Produce the structured report - -```markdown -## Diagnosis: on - -**Values file:** `helm-overrides///custom-values.yaml` -**Namespace:** `` -**Pod state:** `Pending` / `Running on wrong node` / `PVC Pending` / ... - -### Values-side checks -- [✅/⚠️/❌] `nodeSelector` key style matches cluster: `` -- [✅/⚠️/❌] `nodeSelector` value matches an existing pool / ComputeClass on this cluster -- [✅/⚠️/❌] `tolerations` cover the node taints -- [✅/⚠️/❌] `resources.requests` reasonable for cluster's pool sizes -- [✅/⚠️/❌] `persistence.storageClass` (if set) exists in `manifests/storageclass/` -- [✅/⚠️/❌] `existingSecret` (if set) exists per the cluster's `external-secrets/` - -### Cluster-side checks (if accessible) -- Pod events (top 5): - ``` - - ``` -- Node availability summary: - ``` - - ``` - -### Root cause hypothesis - - -### Recommended next step -- **Layer 1 fix (PR):** "Open a PR rewriting `nodeSelector` and `tolerations` in `helm-overrides///custom-values.yaml` from scratch using sibling apps on `` as reference. Specifically: replace `cloud.google.com/compute-class: contour-internal-0-cc` with `dedicated: `." -- **Layer 2 advisory:** (if needed) "Recommend cluster owner increase node-pool max from N to M to relieve resource pressure." -- **Layer 3 refusal:** (if applicable) "Cannot infer the right pool name from this repo alone — needs cluster owner to confirm." - -### References -- [pod-pending-scheduling.md §1](../../docs/platform/runbooks/pod-pending-scheduling.md) -- [contour-nodeselector-tolerations-summary.md](../../contour-nodeselector-tolerations-summary.md) (if Contour) -- [SANCTITY_RULES R5](../../docs/global/SANCTITY_RULES.md) -``` - ---- - -## Output - -A single markdown report (the structured shape above). The skill does **not**: - -- Open a PR (use the matching procedure / `onboard-app` skill). -- Mutate the cluster. -- Click Sync. - -If the recommendation is a Layer 1 PR, name the procedure (e.g. "follow [pod-pending-scheduling §1](../../docs/platform/runbooks/pod-pending-scheduling.md) — values rewrite") rather than blind-generating the diff. - ---- - -## Gotchas - -1. **A pod scheduled-but-on-wrong-node** is harder to diagnose than a Pending pod. Always check `kubectl get pod -o wide` to see the actual node. -2. **PVC `Pending` is sometimes a chained failure** — the pod that would consume it is `Pending` because of scheduling, and the PVC's `WaitForFirstConsumer` mode means it won't bind until a pod is scheduled. Resolve scheduling first. -3. **Some clusters have `taints` that look like keys but are values** — read the actual node label/taint, not the values file's interpretation. -4. **`db-*` dataplane clusters** have minimal sibling apps to compare against. Be extra careful. -5. **Cluster-level issues** (CNI broken, kubelet wedged, node-pool quota) look like scheduling failures from inside the values. Always note the cluster-side context if uncertain. - ---- - -## Layer constraint - -Mostly Layer 2. Output a diagnosis; don't execute. If the diagnosis points at a Layer 1 fix, recommend the matching procedure or `onboard-app` skill; don't open the PR as a side effect. - ---- - -## Related - -- Runbook: [pod-pending-scheduling.md](../../docs/platform/runbooks/pod-pending-scheduling.md). -- Runbook: [ingress-down.md](../../docs/platform/runbooks/ingress-down.md). -- Reference: [contour-nodeselector-tolerations-summary.md](../../contour-nodeselector-tolerations-summary.md). -- Boundaries: [AGENT_BOUNDARIES.md](../../docs/global/AGENT_BOUNDARIES.md). diff --git a/skills/infra/onboard-app.md b/skills/infra/onboard-app.md deleted file mode 100644 index 5aecf73..0000000 --- a/skills/infra/onboard-app.md +++ /dev/null @@ -1,203 +0,0 @@ -# Skill — `onboard-app` - -> **Layer:** Layer 1 — agent generates diff(s) and opens PR(s); humans review and Sync. -> **Scope:** the slice that lives in `devops-infra-helm-charts`. The matching Argo `Application` slice is in `github.com/Meesho/devops-infra-argo-config` and is paired but separate. - -This skill is the agent-callable form of [docs/platform/procedures/onboard-app-to-cluster.md](../../docs/platform/procedures/onboard-app-to-cluster.md). - ---- - -## When to use - -Triggers like: - -- "Onboard `` to ``." -- "Add a `kube-state-metrics` override for `k8s-foo-prd-ase1`." -- "Bring up `external-dns` on the new cluster." - -Do **not** use this skill for: - -- Adding a brand-new cluster (use the `onboard-new-cluster` procedure). -- Bumping a chart's version (use `bump-chart-version`). -- Migrating a chart blue-green (use the `blue-green-chart-migration` procedure directly). - ---- - -## Input - -Required: - -```yaml -app: # must exist in helm-templates/ -target_cluster: # must exist in helm-overrides/ -release_name: # often == app -workload_namespace: # the K8s namespace pods run in -image_tag: # NEVER 'latest' -resources: - cpu_request: - memory_request: - cpu_limit: - memory_limit: -node_pool_key: -node_pool_value: -``` - -Optional: - -```yaml -needs_external_dns: # if true, add a Service to external-dns-services/ -needs_external_secret: # if true, add an ExternalSecret to external-secrets/ -needs_compute_class: # if true (Autopilot), add a ComputeClass under computeclass/ -replicas: -persistence: - enabled: - storage_class: # MUST exist in manifests/storageclass/ - size: -``` - ---- - -## Steps (deterministic) - -1. **Verify pre-conditions.** - ```bash - ls helm-templates//Chart.yaml # chart exists - ls helm-overrides// # cluster exists - ! ls helm-overrides/// # app not already onboarded here - ls manifests/storageclass/.yaml # if persistence.enabled - ``` - -2. **Read the cluster's scheduling profile.** Sample a sibling app on the **same** cluster: - ```bash - ls helm-overrides// | grep -v '^contour' | head -3 - yq e '.nodeSelector' helm-overrides///custom-values.yaml - ``` - Confirm the agent's `node_pool_key` matches the cluster's actual style (`dedicated:` vs `cloud.google.com/compute-class:`). Mismatch → **fail fast** and ask. - -3. **Generate `helm-overrides///custom-values.yaml`.** Use this template: - - ```yaml - image: - registry: asia-southeast1-docker.pkg.dev - repository: meesho-devops-admin-0622/admin/sre/ - tag: - pullPolicy: IfNotPresent - - replicaCount: - - resources: - requests: - cpu: - memory: - limits: - cpu: - memory: - - nodeSelector: - : - tolerations: - - key: - value: - effect: NoSchedule - - # if persistence.enabled - persistence: - enabled: true - storageClass: - size: - accessModes: [ReadWriteOnce] - ``` - -4. **(If `needs_external_dns: true`)** Generate `helm-overrides///external-dns-services/.yaml` per [raw-manifest-sidecar-schema.md §Service for external-dns binding](../../docs/platform/schemas/raw-manifest-sidecar-schema.md). Sample a sibling cluster's pattern. - -5. **(If `needs_external_secret: true`)** Generate an `ExternalSecret` under `helm-overrides//external-secrets/`. Verify the cluster has a `SecretStore` / `ClusterSecretStore` (sample sibling apps' `existingSecret:` references). - -6. **(If `needs_compute_class: true`)** Generate `helm-overrides///computeclass/.yaml`. The `metadata.name` MUST equal ``. - -7. **Validate.** - ```bash - yamllint helm-overrides///custom-values.yaml - helm template helm-templates/ \ - -f helm-overrides///custom-values.yaml > /dev/null - ``` - Render must succeed. If it errors, fail and surface the error. - -8. **Open the PR.** - ```bash - git checkout -b onboard/-on- - git add helm-overrides/// - git commit -m "Onboard to " - git push origin onboard/-on- - gh pr create --base main --title "Onboard to " - ``` - - PR body (use a heredoc): - - ```markdown - ## Summary - Onboards `` to ``. - - - Procedure: `docs/platform/procedures/onboard-app-to-cluster.md` - - Skill: `skills/infra/onboard-app.md` - - Sister-repo PR: - - ## Validation - - `yamllint` passed - - `helm template` rendered cleanly - - Scheduling profile verified against sibling apps on `` - - ## Approvers - - App owner: - - Cluster owner: - - ## CMR - - ``` - ---- - -## Output - -A single PR diff containing 1–3+ new files in this repo: - -- `helm-overrides///custom-values.yaml` -- (optional) `helm-overrides///external-dns-services/.yaml` -- (optional) `helm-overrides///computeclass/.yaml` -- (optional) `helm-overrides//external-secrets/.yaml` - -The skill does **not**: - -- Open the sister-repo PR (separate skill / manual). It must be drafted by the user or a follow-up step. -- Run `argocd app sync` (Layer 1 boundary). -- Modify `helm-templates//`. -- Modify `repository.yaml`. - ---- - -## Pattern reference - -A clean recent onboarding to compare against: pick any merged PR titled "Onboard ... to k8s-...". `git log --oneline --grep='Onboard' -i | head -10`. - ---- - -## Gotchas - -1. **Don't copy a sibling cluster's values verbatim** — per-cluster scheduling differs ([SANCTITY_RULES R5](../../docs/global/SANCTITY_RULES.md)). -2. **Don't set `automated.{prune,selfHeal}` in the sister-repo Application** ([ADR-A5](../../wiki/analyses/ADR-A5-manual-sync-default-for-infra.md)). -3. **Verify `image.tag`** is real before opening the PR — Argo CD only catches a missing tag at sync time. -4. **The release name (`metadata.name` of the resulting `Application`) is set in the sister repo, not here.** Don't assume it; verify with the user. -5. **For Contour**, always cross-reference [contour-nodeselector-tolerations-summary.md](../../contour-nodeselector-tolerations-summary.md) — the matrix is load-bearing. - ---- - -## Layer constraint - -Layer 1. Open the PR; do not merge it; do not Sync. Reviewer + sister-repo PR + Argo CD UI Sync click are the human gates. - ---- - -## Related - -- Procedure: [onboard-app-to-cluster.md](../../docs/platform/procedures/onboard-app-to-cluster.md). -- Schema: [custom-values-schema.md](../../docs/platform/schemas/custom-values-schema.md). -- Boundaries: [AGENT_BOUNDARIES.md](../../docs/global/AGENT_BOUNDARIES.md). diff --git a/wiki/analyses/ADR-A1-cache-vs-upstream-charts.md b/wiki/analyses/ADR-A1-cache-vs-upstream-charts.md deleted file mode 100644 index 29506d5..0000000 --- a/wiki/analyses/ADR-A1-cache-vs-upstream-charts.md +++ /dev/null @@ -1,71 +0,0 @@ -# ADR-A1 — Cache upstream charts in `helm-templates/` (vs. pull on the fly) - -> **Status:** Accepted (de facto — current state of the repo). -> **Repo:** `devops-infra-helm-charts`. -> **Related:** [docs/architecture.md](../../docs/architecture.md), [SANCTITY_RULES R7](../../docs/global/SANCTITY_RULES.md), [update-chart-version.md](../../docs/platform/procedures/update-chart-version.md), [fork-upstream-chart.md](../../docs/platform/procedures/fork-upstream-chart.md). - ---- - -## Context - -Argo CD can render a Helm release in two ways: - -1. **Pull on the fly** — `Application.spec.source.repoURL` points at an upstream Helm registry; `chart:` names the chart. Argo CD pulls the chart at sync time. -2. **Cache locally** — the chart lives in a git repo (this one), and `Application.spec.source.path` points at it. Argo CD reads the chart files from git directly. - -This repo has chosen **caching**. Every chart consumed by Meesho's GKE infra fleet has a directory in `helm-templates//` — usually a thin wrapper whose `Chart.yaml` declares the upstream chart as a dependency, with the resolved subchart materialised in `Chart.lock` and `charts/-.tgz`. - -## Decision - -Cache upstream charts in `helm-templates/` as thin wrapper directories with pinned `dependencies[].version` and a committed `Chart.lock`. Do not configure Argo CD to pull charts directly from upstream registries. - -## Rationale - -1. **Repeatable renders.** A chart-version bump in this repo is a git diff; a chart-version "bump" via upstream pull is whatever the registry returns at sync time. Reproducibility on rollback requires the chart bytes to be in git. - -2. **Air-gapped reviewability.** A reviewer can read the `Chart.yaml`, the `Chart.lock`, and the subchart `.tgz` to know exactly what will render. With on-the-fly pulls, the reviewer trusts the upstream registry hasn't moved a tag. - -3. **Network independence at sync time.** Argo CD's reconcile loop doesn't need outbound network to upstream registries. If GitHub is reachable, sync works; if it isn't, nothing's deploying anyway. - -4. **Supply-chain control.** Pinning `dependencies[].version` plus committing `Chart.lock` means the *digest* of each subchart `.tgz` is recorded. A registry compromise that re-publishes a tag with new content is detected by the lockfile mismatch. - -5. **Forking is local.** When upstream lacks a feature or has a bug, [fork-upstream-chart](../../docs/platform/procedures/fork-upstream-chart.md) is a local edit. No "wait for upstream merge"; the patch lives in our repo until upstream catches up. - -6. **Render-tooling unchanged.** Reviewer-side `helm template helm-templates/ -f overrides.yaml` works locally without any registry config. Pre-merge validation is just `helm template` against the directory. - -## Consequences - -### Accepted - -- **Repo bigger.** 74 chart directories totalling tens of MBs of subchart `.tgz` files. -- **Manual update cadence.** A new upstream release isn't picked up automatically. Someone has to bump `dependencies[].version` and run `helm dependency update`. ([update-chart-version](../../docs/platform/procedures/update-chart-version.md)) -- **Forking risk.** Editing `helm-templates//templates/` casually creates an accidental fork that gets clobbered next `helm dependency update`. ([SANCTITY_RULES R7](../../docs/global/SANCTITY_RULES.md)) -- **Lock-step requirement.** A `Chart.yaml` bump without a refreshed `Chart.lock` is incomplete — Argo CD reads the lockfile, so the version change silently no-ops. - -### Mitigated - -- **Update-chart-version procedure** documents the lock-step requirement explicitly. -- **Versioned siblings** ([ADR-A2](ADR-A2-blue-green-sibling-pattern.md)) handle major-version bumps without losing the old chart. -- **Pre-merge `helm template`** catches values incompatibilities before merge. - -### Open - -- **Stale charts.** Some directories in `helm-templates/` have no current consumer (`grep -rl '' helm-overrides` empty). A periodic clean-up exercise has not been formalised. -- **Subchart `.tgz` size pollution in git history.** Every `helm dependency update` writes a new `.tgz` to `charts/`. Over years, the repo's history grows accordingly. Whether to switch to a Helm-OCI-pull model is an open question. -- **Chart-bump notification.** Nothing currently alerts the team when an upstream advisory affects a chart we have pinned at an old version. Manual diligence today. - -## Alternatives considered - -| Alternative | Why not | -|-------------|---------| -| **Pull-on-the-fly from upstream registries.** | Reproducibility, network dependency, supply-chain risk all worse. | -| **Fully vendor every chart's `templates/`** (no `dependencies[]`, no subchart `.tgz`). | Massive diff churn on every upstream release; worse fork hygiene. | -| **Use Helm OCI registries** as a middle ground (pull `.tgz` from a private registry instead of git). | Plausible Phase-2 work. Adds an extra service to maintain; doesn't solve forking. Not done today. | -| **Use `kustomize` instead of Helm.** | Most upstream charts are Helm; rewriting every chart's templates as Kustomize patches would be enormous. | - -## References - -- Argo CD Helm chart source docs: -- This repo's chart directory layout: [docs/architecture.md §Module boundaries](../../docs/architecture.md). -- Bump procedure: [update-chart-version](../../docs/platform/procedures/update-chart-version.md). -- Fork procedure: [fork-upstream-chart](../../docs/platform/procedures/fork-upstream-chart.md). diff --git a/wiki/analyses/ADR-A2-blue-green-sibling-pattern.md b/wiki/analyses/ADR-A2-blue-green-sibling-pattern.md deleted file mode 100644 index 62da2f9..0000000 --- a/wiki/analyses/ADR-A2-blue-green-sibling-pattern.md +++ /dev/null @@ -1,83 +0,0 @@ -# ADR-A2 — Versioned chart siblings for blue-green migrations - -> **Status:** Accepted (in production — multiple sibling pairs exist today). -> **Repo:** `devops-infra-helm-charts`. -> **Related:** [blue-green-chart-migration.md](../../docs/platform/procedures/blue-green-chart-migration.md), [SANCTITY_RULES R8](../../docs/global/SANCTITY_RULES.md). - ---- - -## Context - -When a chart needs an upgrade with breaking template changes — immutable selector mismatches, removed values keys, major-version semantics — you cannot just bump `dependencies[].version` and call it done. The bump renders a different shape against the same overrides; on every cluster, the next sync would push a Helm upgrade that may fail mid-flight (immutable field) or succeed in ways that surprise the operator. - -The team has chosen a **versioned-sibling** pattern: keep the old chart directory live as `` and introduce the new version as `-` where the variant is one of: - -| Variant | Convention | -|---------|------------| -| `-green` | Blue-green pair (the new is "green") | -| `-vX.Y.Z` | Pinned target version | -| `-latest` | Work-in-progress, soon to subsume the old | -| `-old` | Reverse pattern — `` is the new; `-old` is kept for rollback | - -Live examples in the repo today: - -- `argo-cd` ↔ `argo-cd-green` -- `contour` ↔ `contour-v1.33.3` -- `keda` ↔ `keda-2.17.1` -- `opentelemetry-collector` ↔ `opentelemetry-collector-latest` -- `victoria-metrics-cluster` ↔ `victoria-metrics-cluster-latest` -- `victoria-metrics-agent` ↔ `victoria-metrics-agent-latest` -- `sonarqube` ↔ `sonarqube-old` (reverse — sonarqube is new) - -## Decision - -For chart upgrades that involve breaking changes, create a sibling directory `-` and migrate cluster-by-cluster by repointing the `Application.spec.source.path` in the sister repo. Both directories remain live for the duration of the migration. - -Do not "consolidate" siblings as a maintenance PR — the split is intentional. - -## Rationale - -1. **Per-cluster cutover with rollback.** Each cluster's Argo `Application` flips one path; if the flip fails, that cluster's revert is a one-line PR in the sister repo. Other clusters are untouched. - -2. **No values-shape coupling.** When the new chart uses different values keys, the new chart's overrides can be authored at leisure and tested before any cluster cuts over. The old chart keeps rendering the old shape against the old overrides. - -3. **Immutable-field changes get a clean exit.** A bump in place that changes `Deployment.spec.selector` fails to apply (immutable). The sibling pattern lets you delete-and-recreate the workload as a one-time per-cluster event during cutover, rather than a fleet-wide failure mode. - -4. **Supports staged rollouts.** Some clusters cut over in week 1, others in week 4. The sister repo can hold both states simultaneously without forcing a full-fleet flip. - -5. **Tooling unchanged.** Argo CD, `helm template`, pre-commit hooks all see two parallel directories; no special handling. - -## Consequences - -### Accepted - -- **Repo bigger** during migration windows. A migration in flight has both `` and `-` live. -- **Two charts to maintain** during the window. A CVE patch landing on upstream during the migration may need to be applied to both. -- **Sister repo carries the routing decision.** This repo doesn't know which clusters have cut over; that info lives in `devops-infra-argo-config`. -- **Naming inconsistency.** `-green`, `-vX.Y.Z`, `-latest`, `-old` aren't unified — different migrations chose different conventions. New migrations should pick the most descriptive (`-vX.Y.Z` if the target version is known; `-green` if the migration is a blue-green flip). - -### Mitigated - -- **Procedure** ([blue-green-chart-migration](../../docs/platform/procedures/blue-green-chart-migration.md)) names the steps explicitly: introduce sibling, render-and-diff, per-cluster cutover, retire old. -- **Sanctity rule** ([R8](../../docs/global/SANCTITY_RULES.md)) prevents accidental deletion before all clusters have cut over. - -### Open - -- **Naming convention.** Should the team standardise on `-vX.Y.Z` for all future migrations? Today the choice is ad-hoc. -- **CI/automation** to flag long-running migrations (siblings live > N weeks). Today migrations stall sometimes; nothing alerts. -- **Per-cluster cutover tracking.** Today, knowing "which clusters still point at the old chart" requires `grep` against the sister repo. A small dashboard would help. - -## Alternatives considered - -| Alternative | Why not | -|-------------|---------| -| **In-place bump.** Just change `dependencies[].version` and merge. | Works for compatible bumps; for breaking bumps, fails on the first immutable-field mismatch and may leave the cluster broken. | -| **Branch-as-environment** (e.g. an `int` branch with the new chart). | Doesn't help — Argo CD reads `main`. The sibling pattern is more flexible; per-cluster paths beat branches for this. | -| **Helm `--atomic` upgrades.** Argo can pass `--atomic` to roll back failed upgrades. | Doesn't address breaking values-shape changes that succeed-but-render-wrong. | -| **One big PR that bumps the chart and updates every override.** | Untestable; impossible to roll back one cluster. | - -## References - -- Procedure: [blue-green-chart-migration](../../docs/platform/procedures/blue-green-chart-migration.md). -- Sanctity rule: [R8](../../docs/global/SANCTITY_RULES.md). -- Live siblings (today): [docs/architecture.md §Cross-cutting concerns](../../docs/architecture.md). diff --git a/wiki/analyses/ADR-A3-per-cluster-scheduling.md b/wiki/analyses/ADR-A3-per-cluster-scheduling.md deleted file mode 100644 index d63ba8b..0000000 --- a/wiki/analyses/ADR-A3-per-cluster-scheduling.md +++ /dev/null @@ -1,78 +0,0 @@ -# ADR-A3 — Per-cluster `nodeSelector` / `tolerations` / `computeClass` - -> **Status:** Accepted (status quo — every cluster has bespoke scheduling). -> **Repo:** `devops-infra-helm-charts`. -> **Related:** [contour-nodeselector-tolerations-summary.md](../../contour-nodeselector-tolerations-summary.md), [SANCTITY_RULES R5](../../docs/global/SANCTITY_RULES.md), [pod-pending-scheduling.md](../../docs/platform/runbooks/pod-pending-scheduling.md). - ---- - -## Context - -Meesho's GKE fleet has two cluster types: - -| Type | Scheduling primitives | -|------|------------------------| -| **Standard GKE** | Node pools with `dedicated:` taints and matching node labels | -| **GKE Autopilot** (`k8s-central-prd-ase1`, `k8s-dsgpu-prd-ase1`, `k8s-shared-int-ase1`) | `ComputeClass` resources with `cloud.google.com/compute-class:` keys | - -Within each type, individual clusters have their own node-pool / compute-class topology, designed for the workloads that cluster runs: - -- `k8s-central-mqkafka-prd-ase1` has Kafka-optimised pools. -- `k8s-dsgpu-prd-ase1` has GPU-equipped Autopilot classes. -- `k8s-dengspark-prd-ase1` has Spark-executor pools. -- BU clusters (`k8s-supply-prd-ase1`, `k8s-demand-prd-ase1`, etc.) have per-app pools (`contour-external`, `contour-internal-0`, `monitoring`, …). - -The `helm-overrides///custom-values.yaml` files reflect this — each cluster's values for the same app are different. - -## Decision - -`nodeSelector` / `tolerations` / `affinity` / `topologySpreadConstraints` / `cloud.google.com/compute-class` keys in this repo are **per-cluster, hand-authored, never copied**. The matrix of which Contour instance uses which key on which cluster is recorded in [contour-nodeselector-tolerations-summary.md](../../contour-nodeselector-tolerations-summary.md). For non-Contour apps, sample sibling apps on the same cluster. - -## Rationale - -1. **GKE Autopilot vs Standard isn't optional.** Autopilot's `ComputeClass` mechanism is mutually exclusive with standard `dedicated:` taints. A values block written for one type has no scheduling effect on the other — pods stay `Pending`. - -2. **Per-cluster pool naming is intentional.** `contour-internal-0` on `k8s-supply-prd-ase1` is not the same node pool as `contour-internal-0` on `k8s-demand-prd-ase1` even if they share the name. The pool is sized differently, may have different machine types, may have different anti-affinity rules. Copying values across clusters works *by accident* sometimes; it fails *deliberately* the rest of the time. - -3. **Multi-Contour-per-cluster pattern.** Most BU clusters run 5–6 Contour releases (`contour-external`, `contour-external-1`, `contour-internal-0`, `contour-internal-1`, `contour-internal-intra-{0,1}`). Each has its own pool. Cross-instance copying within the same cluster is also wrong. - -4. **Operational reality.** When a cluster's node pool changes (a new pool added, an old one renamed), only that cluster's overrides need editing. Centralising scheduling values would mean every node-pool change becomes a fleet-wide PR. - -5. **Reviewability.** A reviewer of a values diff can compare against the same file's git history (this cluster's previous state) without needing to know what other clusters look like. Cross-cluster consistency, when it exists, is incidental. - -## Consequences - -### Accepted - -- **The most common silent bug** in this repo is `nodeSelector` / `tolerations` / `computeClass` copied from another cluster. ([SANCTITY_RULES R5](../../docs/global/SANCTITY_RULES.md)) -- **Cross-cluster cleanup is hard.** Renaming a pool (e.g. `monitoring` → `obs-shared`) is N PRs, one per cluster. -- **Onboarding a new cluster** is bespoke per app — every app needs its scheduling block authored from scratch ([onboard-new-cluster](../../docs/platform/procedures/onboard-new-cluster.md)). -- **Reasoning over the fleet** ("which apps are on which pool, fleet-wide?") requires `grep` across cluster directories. - -### Mitigated - -- **The Contour matrix file** ([contour-nodeselector-tolerations-summary.md](../../contour-nodeselector-tolerations-summary.md)) is the single source of truth for the Contour scheduling. **Read before editing any Contour values.** -- **The runbook** ([pod-pending-scheduling.md](../../docs/platform/runbooks/pod-pending-scheduling.md)) walks the diagnosis when scheduling fails. -- **The skill** ([diagnose-scheduling.md](../../skills/infra/diagnose-scheduling.md)) gives an agent a deterministic diagnosis path. - -### Open - -- **A non-Contour scheduling matrix** has not been formalised. Sample-sibling-on-same-cluster is the working approach but isn't written down. -- **Auto-detection of "values copied from another cluster"** is plausible (compare a new file's `nodeSelector` against the cluster's own labels via kubectl). Not implemented. -- **Per-cluster topology drift over time** — when a cluster's underlying pools change in Terraform, the values here need a corresponding update. Today it's manual; ideally a Terraform-side hook would notify. - -## Alternatives considered - -| Alternative | Why not | -|-------------|---------| -| **A shared `_scheduling.yaml`** at the repo root or per-cluster, included via Helm subchart values. | Charts here mostly don't support arbitrary value-file inclusion (Argo CD's `valueFiles` does, but the structure would still need to map per-cluster). Would add a templating step that doesn't exist today. | -| **Centralised "platform values" subchart** that every release inherits. | Requires every chart to be a wrapper that depends on the platform subchart. Most upstream charts aren't structured for this. | -| **Programmatic generation** (a script that emits per-cluster overrides from a topology spec). | Plausible Phase-2 work — the topology spec would need to live somewhere (likely Terraform output), and the generator would need to handle every chart's idiosyncratic values shape. Not done today. | -| **Argo CD `ApplicationSet` with cluster generator + matrix.** | Would centralise routing but doesn't help author the scheduling values. The values still need to be cluster-specific somewhere. | - -## References - -- The matrix: [contour-nodeselector-tolerations-summary.md](../../contour-nodeselector-tolerations-summary.md). -- Sanctity rule: [R5](../../docs/global/SANCTITY_RULES.md). -- Runbook: [pod-pending-scheduling.md](../../docs/platform/runbooks/pod-pending-scheduling.md). -- Skill: [diagnose-scheduling.md](../../skills/infra/diagnose-scheduling.md). diff --git a/wiki/analyses/ADR-A4-raw-manifest-sidecars-in-helm-overrides.md b/wiki/analyses/ADR-A4-raw-manifest-sidecars-in-helm-overrides.md deleted file mode 100644 index 450d4fb..0000000 --- a/wiki/analyses/ADR-A4-raw-manifest-sidecars-in-helm-overrides.md +++ /dev/null @@ -1,75 +0,0 @@ -# ADR-A4 — Raw Kubernetes manifests alongside Helm values in `helm-overrides/` - -> **Status:** Accepted (de facto — the pattern is widespread). -> **Repo:** `devops-infra-helm-charts`. -> **Related:** [docs/platform/schemas/raw-manifest-sidecar-schema.md](../../docs/platform/schemas/raw-manifest-sidecar-schema.md), [docs/architecture.md](../../docs/architecture.md). - ---- - -## Context - -Most directories under `helm-overrides///` contain a single `custom-values.yaml` that Argo CD's Application references via `helm.valueFiles`. But many directories also contain **non-`custom-values` `.yaml` files** that are *not* Helm values. They are raw Kubernetes resources, applied alongside the Helm release by the same Argo Application: - -| Path pattern | Resource kind | -|--------------|---------------| -| `helm-overrides///computeclass/-cc.yaml` | `ComputeClass` (GKE Autopilot) | -| `helm-overrides///external-dns-services/.yaml` | `Service` carrying an `external-dns` annotation | -| `helm-overrides//elastic-cluster/argo-launch.yaml` | `ElasticCluster` (ECK CRD) | -| `helm-overrides///mimir-distributed/alertmanager_config.yaml` | `ConfigMap` materialising Alertmanager config | -| `helm-overrides///external-secrets/*.yaml` (in some shapes) | `ExternalSecret` | - -Argo CD's directory-source mode (`directory.recurse: true` or default flat) walks the whole directory; every `.yaml` file gets applied. The Helm release renders against `custom-values.yaml`; the other files are treated as raw manifests. - -## Decision - -Use a single `helm-overrides///` directory to hold both the Helm values file *and* the raw sidecar manifests an app needs alongside its Helm release. Keep them tightly co-located rather than splitting into separate directories. - -## Rationale - -1. **Atomic deployment unit.** Argo CD applies the directory contents in one Application sync. The Helm release and its supporting `ComputeClass` / `Service` / `ConfigMap` either both appear or neither does — no race between two Applications. - -2. **Reviewer locality.** A PR that "onboards `` on ``" lives in one directory. The reviewer doesn't have to chase across `helm-overrides/`, `manifests/`, and a second sister-repo `Application` to see the full change. - -3. **Argo CD doesn't natively support "Helm + raw manifests" in one source declaratively** — but it does support a directory source that sweeps everything. Co-locating is the pragmatic way to get atomicity. - -4. **Lifecycle coupling.** A `ComputeClass` that an app's `nodeSelector` references is tightly bound to the app — it shouldn't outlive the app, and vice versa. Co-location enforces lifecycle by file proximity. - -5. **Existing CRDs follow the same shape.** ECK's `ElasticCluster`, External Secrets' `ExternalSecret`, Pyroscope's launch manifest — all live next to their app's `custom-values.yaml`. The pattern is consistent. - -## Consequences - -### Accepted - -- **The directory's "shape" is implicit.** Argo CD's behaviour depends on whether the matching `Application` sets `helm.valueFiles` or `directory.recurse`. From inside this repo alone, you can't always tell whether `.yaml` is a sidecar applied alongside Helm, or whether the directory is a raw-only Application that doesn't render Helm. **The matching sister-repo `Application` is the authoritative source.** - -- **Cross-app cleanup is harder.** Removing an app means removing the whole directory; the sidecars come with it. Mostly a feature, occasionally a footgun (a `ConfigMap` that another app references). - -- **Schema overlap risk.** A file named `alertmanager_config.yaml` could be either a values-include or a `ConfigMap` raw manifest. Naming convention matters; review must check. - -- **Cluster-singleton-vs-app-sidecar boundary.** Some resources straddle: a `ComputeClass` is technically cluster-scoped, but it lives under the app that uses it. A `StorageClass` (cluster-scoped, fleet-wide) lives in `manifests/storageclass/` instead. The split between `manifests/` and `helm-overrides///` is "is this resource the app's lifecycle, or is it a long-lived cluster singleton?" — sometimes the answer isn't obvious. - -### Mitigated - -- **Schema doc** ([raw-manifest-sidecar-schema.md](../../docs/platform/schemas/raw-manifest-sidecar-schema.md)) documents the common kinds and the "always pin `apiVersion` and `metadata.namespace`" rule. -- **`manifests/`** is reserved for cluster-wide singletons explicitly, with [storageclass-priorityclass-schema.md](../../docs/platform/schemas/storageclass-priorityclass-schema.md) documenting the boundary. - -### Open - -- **No formal indicator in this repo** of whether a given directory is "Helm + sidecars" or "raw only." The user has to read the sister-repo `Application` to know. -- **Naming for sub-directories** (`computeclass/`, `external-dns-services/`, `external-secrets/`) is conventional but not enforced. New patterns get added ad-hoc. -- **Some `manifests/` content arguably should be in `helm-overrides///`** (e.g. per-cluster Jenkins filestore PV/PVCs are tied to a Jenkins release). The current split was historical; revisiting it is open. - -## Alternatives considered - -| Alternative | Why not | -|-------------|---------| -| **Two Argo Applications per app — one Helm, one raw.** | Loses atomicity; introduces sync-ordering races. | -| **Render every sidecar through Helm** by inlining it as a `templates/` file in a forked chart. | Forks a chart we'd otherwise leave vanilla; conflicts with [ADR-A1](ADR-A1-cache-vs-upstream-charts.md). | -| **Move sidecars into a separate `cluster-resources//` tree.** | Loses lifecycle coupling; a separate directory tree to maintain. Reviewer must cross-reference. | -| **Use Helm's post-renderer hooks** to inject sidecars into the Helm release. | Adds tooling complexity; doesn't help when the sidecar is a different `apiVersion` than the chart understands. | - -## References - -- Schema: [raw-manifest-sidecar-schema.md](../../docs/platform/schemas/raw-manifest-sidecar-schema.md). -- Schema: [storageclass-priorityclass-schema.md](../../docs/platform/schemas/storageclass-priorityclass-schema.md) — for the `manifests/` boundary. -- Procedure: [onboard-app-to-cluster.md](../../docs/platform/procedures/onboard-app-to-cluster.md). diff --git a/wiki/analyses/ADR-A5-manual-sync-default-for-infra.md b/wiki/analyses/ADR-A5-manual-sync-default-for-infra.md deleted file mode 100644 index 3ad6acb..0000000 --- a/wiki/analyses/ADR-A5-manual-sync-default-for-infra.md +++ /dev/null @@ -1,66 +0,0 @@ -# ADR-A5 — Manual sync is the default for infra Applications - -> **Status:** Accepted (status quo — most infra Applications lack `automated`). -> **Repo:** `devops-infra-helm-charts` (consumer of the decision; the `Application` shapes that enact it live in the sister repo). -> **Related:** [SANCTITY_RULES R3](../../docs/global/SANCTITY_RULES.md) (analogue from the application-side repo), [argocd-sync-failure.md](../../docs/platform/runbooks/argocd-sync-failure.md), [deboard-app.md](../../docs/platform/procedures/deboard-app.md). - ---- - -## Context - -Argo CD `Application` resources can have a `spec.syncPolicy.automated` block that auto-applies any diff between git and the cluster on every reconcile cycle. With it: a merge to `main` deploys immediately. Without it: a merge updates the Application's *desired state*, but the cluster doesn't change until a human (or external trigger) clicks **Sync** in the Argo CD UI (or runs `argocd app sync`). - -Most Applications routing to this repo (`devops-infra-helm-charts`) **do not have `automated`** set. A small minority of infra Applications — typically things that should self-heal aggressively (canary-bot, statsd-exporter, vmextractor) — do. - -## Decision - -For infra Applications routed by `github.com/Meesho/devops-infra-argo-config`, the default is **manual sync** — `spec.syncPolicy` contains only `syncOptions: [CreateNamespace=true]`, with no `automated` block. Adding `automated.{prune,selfHeal}: true` to a service-tier Application is a deliberate, headline-of-the-PR change. - -## Rationale - -1. **Production blast radius.** A merge here can cascade across many clusters. If a values change is wrong, the manual-sync default means an operator has a chance to spot it (in Argo CD's Diff view) before clicking through. Auto-sync would push the broken change to every cluster simultaneously on the next reconcile. - -2. **Per-cluster cutover.** A typical chart bump or values change rolls out cluster-by-cluster. The operator clicks Sync on cluster A, watches, then proceeds to cluster B. Auto-sync forces a fleet-wide flip with no soak window. - -3. **Out-of-band drift detection.** Manual sync makes drift visible — when someone `kubectl edit`-s a release on a cluster, Argo CD shows `OutOfSync` and surfaces the diff. With auto-sync, the drift is silently overwritten on the next reconcile, hiding the fact that someone made an out-of-band change. - -4. **Sync click is the agent's hard stop.** [AGENT_BOUNDARIES.md](../../docs/global/AGENT_BOUNDARIES.md) classifies "click Sync" as Layer 2 advisory — the agent recommends the command but never executes. The default of manual sync makes this enforceable: the agent literally cannot deploy without a human in the loop. - -5. **Safe by default; opt in for the loop closures.** Apps that genuinely should self-heal (canary-bot — purpose is to test traffic; statsd-exporter — purpose is fleet-wide telemetry) can be opted in via `automated.prune: true`. The opt-in is a deliberate decision, not a side-effect. - -## Consequences - -### Accepted - -- **Operational tax.** Every PR merge creates an `OutOfSync` Application that someone has to click through. With ~30 clusters × dozens of apps, this can pile up on busy days. -- **Drift accumulation.** A PR that nobody clicks Sync on sits as `OutOfSync` indefinitely. Sometimes this is intentional (the PR was speculative); sometimes it's forgotten. Periodic audits ("which Applications have been `OutOfSync` for > 7 days?") aren't yet automated. -- **Manual-sync bias** can mask incidents — an Application failing to sync (because of a values regression) may sit `OutOfSync` for a while before someone notices. Auto-sync would have surfaced it loudly via failed reconciles. - -### Mitigated - -- **The runbook** ([argocd-sync-failure.md](../../docs/platform/runbooks/argocd-sync-failure.md)) explicitly handles the "OutOfSync only, no error, sync hasn't run" branch as §4 — its own diagnostic path. -- **Sanctity rule analogue** in the application-side `devops-argo-config` repo names this explicitly (R3); we inherit the principle. -- **Skill** ([diagnose-scheduling.md](../../skills/infra/diagnose-scheduling.md)) outputs Layer 2 advisories ("recommend operator clicks Sync") rather than auto-Sync triggers. - -### Open - -- **Notification on long-`OutOfSync` Applications** — periodic alert / dashboard. Today operators just see this in the Argo UI. -- **Should some infra apps move to auto-sync?** Specifically, sidecars whose blast radius is tiny (telemetry collectors, log agents). A periodic review hasn't been done. -- **The opt-in list of currently auto-synced apps** isn't documented in this repo. Has to be inferred from the sister repo's `Application` files. -- **Cluster-specific opt-ins** — auto-sync on dev clusters but manual on prod — would be a reasonable refinement but adds per-cluster `Application` divergence. - -## Alternatives considered - -| Alternative | Why not | -|-------------|---------| -| **Auto-sync everywhere by default.** | Loses the per-cluster operator gate; a bad merge cascades fleet-wide. | -| **Auto-sync with `selfHeal: false` but `prune: true`.** | Still applies values changes immediately; doesn't help. | -| **Manual sync but with auto-fallback after N hours.** | Argo CD doesn't offer this natively. Building it would require a controller. | -| **Per-environment policy** (auto-sync on int, manual on prod). | Reasonable refinement; adds policy state to the sister repo. Could be future work. | - -## References - -- Sister-repo `Application` shapes: `github.com/Meesho/devops-infra-argo-config`. -- Runbook §4: [argocd-sync-failure.md](../../docs/platform/runbooks/argocd-sync-failure.md). -- Boundaries: [AGENT_BOUNDARIES.md](../../docs/global/AGENT_BOUNDARIES.md) (Layer 2 row "Sync `` now in Argo CD"). -- Application-side analogue: `devops-argo-config`'s `SANCTITY_RULES.md R3`. diff --git a/wiki/entities/DevOps Infra Helm Charts.md b/wiki/entities/DevOps Infra Helm Charts.md deleted file mode 100644 index af8d003..0000000 --- a/wiki/entities/DevOps Infra Helm Charts.md +++ /dev/null @@ -1,214 +0,0 @@ -# Wiki Entity — DevOps Infra Helm Charts - -> Architectural reference page for `devops-infra-helm-charts`. Operational depth lives in the procedures and runbooks under `docs/platform/`. This page is what you cite from other wiki pages when you mean "the infra Helm values repo." - ---- - -## What it is - -The single GitOps source-of-truth for **what infrastructure tooling runs on Meesho's GKE fleet, where, and with what values**. Sister repo `github.com/Meesho/devops-infra-argo-config` is the routing layer — it holds the Argo CD `Application` / `ApplicationSet` manifests that point at paths in this repo. - -A merge to `main` is a deploy event: Argo CD on each cluster reconciles from `main`. **Most infra Applications use manual sync** ([ADR-A5](../analyses/ADR-A5-manual-sync-default-for-infra.md)), so a merge updates the Application resource but a human Sync click on the cluster's Argo CD UI deploys the workload. - -There is no application code, no build, no tests — only declarative YAML (Helm charts, values overrides, Kubernetes manifests) and two git-hook shell scripts. - ---- - -## What it controls - -| Surface | Count | -|---------|-------| -| Cached / forked upstream Helm charts in `helm-templates/` | **74** | -| Cluster directories under `helm-overrides/` | 30+ (16 BU prod + 5 GCP twins + dataplane `db-*` + int + dev + Aurva) | -| Cluster-wide singletons under `manifests/` | StorageClasses (4), per-cluster PriorityClasses, Jenkins/JFrog filestore PV/PVCs | -| Active pre-commit hooks | 1 (TruffleHog) | -| No-op pre-commit hooks (gated paths absent) | 2 (CAC, Yaak) | -| Post-commit hooks | 1 (Cursor metric collector — non-blocking) | - ---- - -## Architecture - -### Two-repo GitOps split - -```text -┌─────────────────────────────────┐ ┌──────────────────────────────────┐ -│ devops-infra-helm-charts │ │ devops-infra-argo-config │ -│ (this repo — values + charts) │ ◄───── │ (sister repo — routing) │ -│ │ path: │ │ -│ helm-templates// │ │ Application / ApplicationSet │ -│ helm-overrides///│ │ spec.source.path: │ -│ manifests// │ │ helm-overrides/<...> │ -└─────────────────────────────────┘ └──────────────────────────────────┘ - │ - ▼ - ┌──────────────────────────────┐ - │ Per-cluster Argo CD │ - │ (one per workload cluster) │ - │ reconciles main → cluster │ - └──────────────────────────────┘ -``` - -### Deploy lifecycle - -```text -edit helm-overrides///custom-values.yaml - │ - ▼ -git commit ──► pre-commit hook (TruffleHog secret scan) - │ - ▼ -git push ──► PR → review → merge to main - │ - ▼ -Argo CD on each cluster reconciles main + sister-repo main - │ - ▼ -Application sync: helm template -f → apply (manual Sync click for most) - │ - ▼ -post-commit hook ships Cursor AI metrics (background, non-blocking) -``` - ---- - -## Cluster fleet - -All in `asia-southeast1` (zone-a or zone-c), fleet `meesho-admin-prd-0622`. See [docs/architecture.md](../../docs/architecture.md) for the full inventory. - -### BU prod clusters (`k8s--prd-ase1[c]`) - -| Cluster type | Examples | -|--------------|----------| -| Standard GKE prod | `k8s-supply-prd-ase1`, `k8s-demand-prd-ase1`, `k8s-dataengg-prd-ase1`, `k8s-datascience-prd-ase1`, `k8s-farmiso-prd-ase1`, `k8s-ml-platform-prd-ase1`, `k8s-admin-prd-ase1`, `k8s-sec-admin-ase1`, `k8s-devops-admin-ase1` | -| GKE Autopilot | `k8s-central-prd-ase1`, `k8s-dsgpu-prd-ase1`, `k8s-shared-int-ase1` | -| Specialty | `k8s-central-mqkafka-prd-ase1`, `k8s-dengspark-prd-ase1`, `k8s-dengspark-di-prd-ase1`, `k8s-dengspark-notebook-prd-ase1`, `k8s-dscispark-prd-ase1` | -| GCP zone-c twins | `k8s-supply-prd-ase1c`, `k8s-demand-prd-ase1c`, `k8s-dataengg-prd-ase1c`, `k8s-datascience-prd-ase1c`, `k8s-central-prd-ase1c` | - -### Other clusters - -| Pattern | Use | -|---------|-----| -| `k8s-shared-int-ase1` | Shared int (pre-prod) — only non-prod BU cluster | -| `k8s-aurva-prd-ase1` | Aurva integration (minimal override set) | -| `k8s-supply-dev-ase1` | Dev/sandbox supply | -| `db--...` | Auto-named dataplane clusters (minimal: `kube-state-metrics` + `victoria-metrics-agent`) | - ---- - -## Chart inventory - -Categorised view; full list in [docs/architecture.md §Helm chart inventory](../../docs/architecture.md). - -| Category | Charts | -|----------|--------| -| Argo / GitOps | `argo-cd`, `argo-cd-green` | -| Ingress / edge | `contour`, `contour-v1.33.3`, `contour-ca-issuer`, `contour-cert-checker`, `ingress-nginx`, `cert-manager`, `external-dns`, `external-secrets` | -| Observability — metrics | `prometheus-node-exporter`, `prometheus-stackdriver-exporter`, `kube-state-metrics`, `kube-events`, `victoria-metrics-{single,cluster,cluster-latest,agent,agent-latest,alert,alert-stateful,alerts-config,auth,mcp}`, `vm-alert-config`, `mimir-distributed`, `pmm`, `telegraf-operator` | -| Observability — logs/traces/profiles | `fluentd`, `loki-distributed`, `tempo-distributed`, `pyroscope`, `alloy`, `opentelemetry-collector`, `opentelemetry-collector-latest`, `opentelemetry-operator`, `elastalert2`, `coroot-node-agent`, `deepfence-console`, `deepfence-router` | -| UI / dashboards | `grafana`, `grafana-edge`, `grafana-mcp`, `kubernetes-dashboard`, `superset`, `uptime-kuma` | -| Workflow / CI/CD | `jenkins`, `jfrog`, `sonarqube`, `sonarqube-old`, `flagger`, `keda`, `keda-2.17.1`, `kyverno`, `loadtester`, `temporal`, `dind`, `canary-bot-gcp`, `paused-container` | -| Networking / DNS | `coredns`, `kube-dns`, `bifrost`, `conntrack-adjuster`, `node-thp-config` | -| Data / search / DB | `clickhouse`, `etcd`, `vault`, `elasticsearch-mcp`, `eck-operator`, `athens-proxy` | -| AI / 3rd-party | `aurva-dataplane`, `deepgram-onprem`, `rancher` | - -Versioned siblings (blue-green migration targets) are intentional, not duplicates — see [ADR-A2](../analyses/ADR-A2-blue-green-sibling-pattern.md). - ---- - -## Upstreams (what this repo *needs*) - -| Upstream | Why we need it | -|----------|----------------| -| Sister repo `devops-infra-argo-config` | The routing layer. Without an `Application` / `ApplicationSet` over there, paths here are inert. | -| Per-cluster Argo CD instances | Reconcile main into each cluster. Bootstrapping lives outside this repo. | -| GCP Artifact Registry mirror (`asia-southeast1-docker.pkg.dev/meesho-devops-admin-0622/admin/sre/`) | All production image tags resolve here. | -| External Secrets Operator (per-cluster `external-secrets` app) | Materialises GCP Secret Manager / Vault secrets into K8s `Secret`s referenced by `existingSecret:` keys. | -| TruffleHog webhook (`observe.meeshogcp.in/api/webhook`) | Pre-commit secret-scan telemetry. | -| Cursor metric API (`cursor-server.meeshogcp.in/api/v1/...`) | Post-commit (non-blocking) Cursor AI usage metrics. | -| `cicd-scripts` repo | Source of pre/post-commit hook script logic. | -| `registry-bootstrap` automation | Owns `repository.yaml`. | - -## Downstreams (what depends on this repo) - -| Downstream | Failure mode if this repo is broken | -|------------|--------------------------------------| -| Per-cluster Argo CD | If a chart's `Chart.lock` is missing or the values don't render, that cluster's Argo Application reports sync failure. | -| Every infra workload (Contour, VictoriaMetrics, Argo CD, cert-manager, …) | A bad values change can take down ingress, observability, secret materialisation. | -| Pulse alerting / on-call routing | Reads label metadata on alerts; chart-bump-induced label drift can break routing. | -| `external-dns` / Cloud DNS | Sidecar `Service` resources here drive DNS records. | - ---- - -## Key conventions (load-bearing) - -| Convention | What enforces it | -|------------|------------------| -| `helm-overrides///custom-values.yaml` is the values filename | Sister-repo `Application.spec.source.helm.valueFiles` references this name | -| Cluster directory name == cluster name in Argo CD | Convention only — silent bind failure if mismatched | -| `image.registry: asia-southeast1-docker.pkg.dev` | [SANCTITY_RULES R11](../../docs/global/SANCTITY_RULES.md) | -| `nodeSelector` / `tolerations` per-cluster bespoke | [SANCTITY_RULES R5](../../docs/global/SANCTITY_RULES.md), [contour-nodeselector-tolerations-summary.md](../../contour-nodeselector-tolerations-summary.md) | -| `fullnameOverride` is stable forever | [SANCTITY_RULES R9](../../docs/global/SANCTITY_RULES.md) | -| Versioned chart siblings stay live during migrations | [SANCTITY_RULES R8](../../docs/global/SANCTITY_RULES.md), [ADR-A2](../analyses/ADR-A2-blue-green-sibling-pattern.md) | -| `helm-templates//templates/` is upstream — don't edit casually | [SANCTITY_RULES R7](../../docs/global/SANCTITY_RULES.md), [ADR-A1](../analyses/ADR-A1-cache-vs-upstream-charts.md) | - ---- - -## Operational procedures - -| Task | Procedure | -|------|-----------| -| Onboard an app to a cluster | [onboard-app-to-cluster](../../docs/platform/procedures/onboard-app-to-cluster.md) | -| Onboard a brand-new cluster's overrides | [onboard-new-cluster](../../docs/platform/procedures/onboard-new-cluster.md) | -| Bump a chart's pinned version | [update-chart-version](../../docs/platform/procedures/update-chart-version.md) | -| Intentionally fork a chart | [fork-upstream-chart](../../docs/platform/procedures/fork-upstream-chart.md) | -| Migrate a chart blue-green | [blue-green-chart-migration](../../docs/platform/procedures/blue-green-chart-migration.md) | -| Deboard a retired app | [deboard-app](../../docs/platform/procedures/deboard-app.md) | - -## Runbooks - -| Symptom | Runbook | -|---------|---------| -| Argo CD app errored / OutOfSync | [argocd-sync-failure](../../docs/platform/runbooks/argocd-sync-failure.md) | -| Ingress (Contour) is down | [ingress-down](../../docs/platform/runbooks/ingress-down.md) | -| Pods Pending / wrong-node scheduling | [pod-pending-scheduling](../../docs/platform/runbooks/pod-pending-scheduling.md) | - ---- - -## Architecture decisions - -| ADR | Decision | -|-----|----------| -| [ADR-A1](../analyses/ADR-A1-cache-vs-upstream-charts.md) | Why we cache upstream charts in `helm-templates/` instead of pulling on the fly | -| [ADR-A2](../analyses/ADR-A2-blue-green-sibling-pattern.md) | Why we use versioned chart siblings for migrations | -| [ADR-A3](../analyses/ADR-A3-per-cluster-scheduling.md) | Why per-cluster `nodeSelector` / `tolerations` / `computeClass` is bespoke | -| [ADR-A4](../analyses/ADR-A4-raw-manifest-sidecars-in-helm-overrides.md) | Why `helm-overrides///` mixes Helm values with raw sidecar manifests | -| [ADR-A5](../analyses/ADR-A5-manual-sync-default-for-infra.md) | Why most infra Applications are manual-sync (no `automated`) | - ---- - -## Open knowledge gaps - -1. **The bootstrap source for per-cluster Argo CD installs is outside this repo.** Likely Terraform-managed cluster config or a separate "argo-bootstrap" repo. Locating and documenting it is a follow-up. -2. **Some `helm-templates//` charts have no consumers** (`grep -rl '' helm-overrides` returns empty). Inventory and cleanup is a separate exercise. -3. **The split between `helm-overrides///.yaml` raw sidecars and pure-Helm values directories isn't formally documented per app.** The matching sister-repo `Application` is authoritative; this repo doesn't always make the shape obvious from a glance. -4. **Cross-cluster project-replica pattern** (e.g. mrouter-equivalent for infra). Doesn't exist here in the same way it does in `devops-argo-config`, but the dataplane (`db-*`) clusters do share a structure that could be templated. - ---- - -## Ownership - -- **Primary**: `siddharth.pal@meesho.com` (per `repository.yaml`) -- **Secondary**: `samarth.nag@meesho.com` -- **Team**: DevOps / Platform - ---- - -## Related wiki entities - -- `[[DevOps Infra Argo Config]]` — sister repo (Argo `Application` / `ApplicationSet` routing). -- `[[DevOps ArgoCD Config]]` — application-side GitOps (services, not infra). -- `[[DevOps Helm Charts]]` — the *application*-side chart repo (services), distinct from this one. -- `[[CI-CD Security Tools]]` / `[[Git Hooks Security Pipeline]]` — pre-commit hook source. -- `[[Per-Cluster Deployment Contract]]` — the cross-repo contract for bringing up an app on a cluster. -- `[[GitOps with ArgoCD]]` — overarching GitOps concept page.