From b4484893cabe2e8f50bb7df5384143f3c0d3b36d Mon Sep 17 00:00:00 2001 From: Outlier Date: Sat, 1 Aug 2026 23:40:41 +0300 Subject: [PATCH] feat(monitor): add Prometheus, Grafana, exporters, GPU scrape and cluster dashboard --- docs/MONITORING.md | 85 ++++++++ docs/README.md | 1 + infra/k8s/apps/monitor.yaml | 22 ++ .../dashboards/deephorizon-cluster.json | 133 ++++++++++++ infra/k8s/monitor/grafana.yaml | 133 ++++++++++++ infra/k8s/monitor/kube-state-metrics.yaml | 94 ++++++++ infra/k8s/monitor/kustomization.yaml | 19 ++ infra/k8s/monitor/namespace.yaml | 4 + infra/k8s/monitor/node-exporter.yaml | 77 +++++++ infra/k8s/monitor/prometheus.yaml | 205 ++++++++++++++++++ 10 files changed, 773 insertions(+) create mode 100644 docs/MONITORING.md create mode 100644 infra/k8s/apps/monitor.yaml create mode 100644 infra/k8s/monitor/dashboards/deephorizon-cluster.json create mode 100644 infra/k8s/monitor/grafana.yaml create mode 100644 infra/k8s/monitor/kube-state-metrics.yaml create mode 100644 infra/k8s/monitor/kustomization.yaml create mode 100644 infra/k8s/monitor/namespace.yaml create mode 100644 infra/k8s/monitor/node-exporter.yaml create mode 100644 infra/k8s/monitor/prometheus.yaml diff --git a/docs/MONITORING.md b/docs/MONITORING.md new file mode 100644 index 0000000..3987156 --- /dev/null +++ b/docs/MONITORING.md @@ -0,0 +1,85 @@ +# Monitoring — Prometheus + Grafana + +Cluster, node ve **GPU** metriklerini toplar (Prometheus), gösterir (Grafana). +Kurulum GitOps ile: `infra/k8s/monitor/`, `apps/monitor.yaml`, namespace +`deephorizon-monitor`. + +## Erişim + +| | | +|:---|:---| +| **Grafana (LAN)** | `http://10.10.1.132:30030` | +| **Kullanıcı** | `admin` — parola DevOps'ta (parola kasası) | +| **Prometheus** | ClusterIP (dışarı açılmaz); Grafana'ya datasource olarak önceden bağlı | + +UFW 30030'u yalnızca yerel subnet'e açar. İleride NPM arkasına domain'le alınabilir. + +## Ne izler + +| Kaynak | Ne verir | +|:---|:---| +| **node-exporter** | node CPU / RAM / disk / ağ | +| **kube-state-metrics** | pod/deployment/PVC/job durumları (restart, hazır replika…) | +| **cAdvisor** (kubelet) | konteyner kaynak kullanımı | +| **dcgm-exporter** (GPU operator) | L40S kullanım / VRAM / sıcaklık / güç | +| **anotasyon keşfi** | `prometheus.io/scrape: "true"` taşıyan her pod (api, inference) | + +> dcgm-exporter anotasyon taşımadığı için özel bir scrape job'ı ile toplanır +> (`gpu-operator-resources/nvidia-dcgm-exporter`). api/inference deploy olunca +> anotasyonla otomatik girer — ekstra iş yok. + +## Dashboard'lar + +- **DeepHorizon - Cluster** — koddan otomatik gelir (`infra/k8s/monitor/dashboards/deephorizon-cluster.json`, + Grafana provisioning). Pod/CPU/RAM/restart + deployment durumu, bizim etiket şemamıza göre. +- **GPU + Node** — topluluk panoları, repoya gömülmedi (JSON büyük). Deploy sonrası + **elle import** (Grafana → New → Import → ID → datasource `Prometheus`): + - **12239** — NVIDIA DCGM (GPU: kullanım, VRAM, sıcaklık, güç) + - **1860** — Node Exporter Full (CPU / RAM / disk / ağ) + +## Grafana admin parolası — elle (Git'e girmez) + +Grafana pod'u `grafana-admin` Secret'ı olmadan başlamaz. Secret'lar Git dışında; +SealedSecret ile bir kez uygulanır: + +```bash +# 1. Duz secret taslagi (commit etme) +kubectl create secret generic grafana-admin \ + --namespace deephorizon-monitor \ + --from-literal=admin-password='' \ + --dry-run=client -o yaml > /tmp/grafana-admin.yaml + +# 2. Muhurle (namespace scope zorunlu) +kubeseal -n deephorizon-monitor -o yaml \ + < /tmp/grafana-admin.yaml > /tmp/grafana-admin-sealed.yaml + +# 3. Cluster'a uygula (Git'e DEGIL) +kubectl apply -f /tmp/grafana-admin-sealed.yaml + +# 4. Temizle; muhurlu YAML'i parola kasasina koy +shred -u /tmp/grafana-admin.yaml /tmp/grafana-admin-sealed.yaml +``` + +> **Sync sırası:** Grafana bu secret'a bağlı. Argo CD namespace'i oluşturduktan +> sonra secret'ı apply et; gelene kadar Grafana `CreateContainerConfigError`'da +> bekler, secret gelince kendi düzelir (takılırsa `kubectl delete pod -n +> deephorizon-monitor -l app=grafana`). + +## DevOps notları + +| | | +|:---|:---| +| Manifest'ler | `infra/k8s/monitor/` (kustomize) | +| Argo CD app | `infra/k8s/apps/monitor.yaml` | +| Grafana NodePort | `30030` (UFW: `ufw allow from 10.10.1.0/24 to any port 30030 proto tcp`) | +| Secret | `grafana-admin` (SealedSecret, Git dışı) | +| Dashboard'lar | `infra/k8s/monitor/dashboards/*.json` → `grafana-dashboards` ConfigMap (generator) | + +**Yeni scrape hedefi eklemek:** pod'a `prometheus.io/scrape: "true"` + +`prometheus.io/port: ""` anotasyonu koy → otomatik toplanır. Anotasyon +konamıyorsa (üçüncü parti) `prometheus.yaml`'e özel bir job ekle (dcgm örneği gibi), +yeniden render/apply et. + +**Yeni dashboard eklemek:** JSON'ı `infra/k8s/monitor/dashboards/`'a koy → +`kustomization.yaml`'deki `configMapGenerator.files` listesine ekle → Grafana +provider onu otomatik yükler. diff --git a/docs/README.md b/docs/README.md index be3282d..eb9d9f1 100644 --- a/docs/README.md +++ b/docs/README.md @@ -11,6 +11,7 @@ Project documentation that does not belong in source-level READMEs. | [`DVC.md`](DVC.md) | **DVC rehberi (TR)** — veri versiyonlama: `dvc-cache` remote'unun kurulumu (`dvc init` + `remote add`), push/pull kullanımı, eski sürüme dönme. **DVC kullanacak kişinin adresi.** | | [`AIRFLOW.md`](AIRFLOW.md) | **Airflow rehberi (TR)** — veri pipeline orkestrasyonu: erişim, DAG teslimi (git-sync), çalışma ortamı kontratı (env, workspace, imaj), Data squad'dan kalanlar. **DAG yazacak kişinin adresi.** | | [`DEVOPS.md`](DEVOPS.md) | **DevOps kurulum günlüğü (TR)** — GPU sunucu bootstrap'ının tamamı: NVIDIA sürücü, MicroK8s + GPU addon, Sealed Secrets, Argo CD, NPM, UFW; karşılaşılan hatalar, sebepleri ve çözümleri. | +| [`MONITORING.md`](MONITORING.md) | **Monitoring rehberi (TR)** — Prometheus + Grafana + exporter'lar + GPU (dcgm): erişim, ne izlenir, dashboard'lar, `grafana-admin` SealedSecret akışı, yeni scrape/dashboard ekleme. | | `adr/` | Architecture Decision Records — one file per non-trivial decision | | `runbooks/` | On-call / incident playbooks (one per failure mode) | diff --git a/infra/k8s/apps/monitor.yaml b/infra/k8s/apps/monitor.yaml new file mode 100644 index 0000000..a020559 --- /dev/null +++ b/infra/k8s/apps/monitor.yaml @@ -0,0 +1,22 @@ +# Monitoring stack (kustomize). Grafana admin parolasi Git'te degil: +# 'grafana-admin' SealedSecret'i elle apply edilmeli (bkz. docs/MONITORING.md). +apiVersion: argoproj.io/v1alpha1 +kind: Application +metadata: + name: monitor + namespace: argocd +spec: + project: default + source: + repoURL: https://github.com/Octapull/deephorizon.git + targetRevision: main + path: infra/k8s/monitor + destination: + server: https://kubernetes.default.svc + namespace: deephorizon-monitor + syncPolicy: + automated: + prune: true + selfHeal: true + syncOptions: + - CreateNamespace=true diff --git a/infra/k8s/monitor/dashboards/deephorizon-cluster.json b/infra/k8s/monitor/dashboards/deephorizon-cluster.json new file mode 100644 index 0000000..fdaeda5 --- /dev/null +++ b/infra/k8s/monitor/dashboards/deephorizon-cluster.json @@ -0,0 +1,133 @@ +{ + "title": "DeepHorizon - Cluster", + "uid": "deephorizon-cluster", + "schemaVersion": 39, + "version": 1, + "editable": true, + "time": { "from": "now-6h", "to": "now" }, + "refresh": "30s", + "templating": { + "list": [ + { + "name": "datasource", + "type": "datasource", + "query": "prometheus", + "current": {}, + "hide": 0 + }, + { + "name": "namespace", + "type": "query", + "datasource": { "type": "prometheus", "uid": "${datasource}" }, + "query": "label_values(kube_pod_info, namespace)", + "definition": "label_values(kube_pod_info, namespace)", + "refresh": 2, + "includeAll": true, + "multi": true, + "current": { "text": "All", "value": "$__all" } + } + ] + }, + "panels": [ + { + "id": 1, + "type": "stat", + "title": "Running Pods", + "gridPos": { "h": 4, "w": 6, "x": 0, "y": 0 }, + "datasource": { "type": "prometheus", "uid": "${datasource}" }, + "options": { "colorMode": "value", "graphMode": "area", "reduceOptions": { "calcs": ["lastNotNull"] } }, + "targets": [ + { "refId": "A", "datasource": { "type": "prometheus", "uid": "${datasource}" }, + "expr": "sum(kube_pod_status_phase{phase=\"Running\"})" } + ] + }, + { + "id": 2, + "type": "stat", + "title": "Pods Not Running", + "gridPos": { "h": 4, "w": 6, "x": 6, "y": 0 }, + "datasource": { "type": "prometheus", "uid": "${datasource}" }, + "fieldConfig": { "defaults": { "thresholds": { "mode": "absolute", "steps": [ { "color": "green", "value": null }, { "color": "red", "value": 1 } ] } }, "overrides": [] }, + "options": { "colorMode": "value", "graphMode": "none", "reduceOptions": { "calcs": ["lastNotNull"] } }, + "targets": [ + { "refId": "A", "datasource": { "type": "prometheus", "uid": "${datasource}" }, + "expr": "sum(kube_pod_status_phase{phase!=\"Running\"})" } + ] + }, + { + "id": 3, + "type": "stat", + "title": "Container Restarts (total)", + "gridPos": { "h": 4, "w": 6, "x": 12, "y": 0 }, + "datasource": { "type": "prometheus", "uid": "${datasource}" }, + "options": { "colorMode": "value", "graphMode": "area", "reduceOptions": { "calcs": ["lastNotNull"] } }, + "targets": [ + { "refId": "A", "datasource": { "type": "prometheus", "uid": "${datasource}" }, + "expr": "sum(kube_pod_container_status_restarts_total)" } + ] + }, + { + "id": 4, + "type": "stat", + "title": "Nodes Ready", + "gridPos": { "h": 4, "w": 6, "x": 18, "y": 0 }, + "datasource": { "type": "prometheus", "uid": "${datasource}" }, + "options": { "colorMode": "value", "graphMode": "none", "reduceOptions": { "calcs": ["lastNotNull"] } }, + "targets": [ + { "refId": "A", "datasource": { "type": "prometheus", "uid": "${datasource}" }, + "expr": "sum(kube_node_status_condition{condition=\"Ready\",status=\"true\"})" } + ] + }, + { + "id": 5, + "type": "timeseries", + "title": "CPU by pod (cores, top 10)", + "gridPos": { "h": 8, "w": 12, "x": 0, "y": 4 }, + "datasource": { "type": "prometheus", "uid": "${datasource}" }, + "fieldConfig": { "defaults": { "unit": "short" }, "overrides": [] }, + "targets": [ + { "refId": "A", "datasource": { "type": "prometheus", "uid": "${datasource}" }, + "legendFormat": "{{namespace}}/{{pod}}", + "expr": "topk(10, sum by (namespace, pod) (rate(container_cpu_usage_seconds_total{container!=\"\", namespace=~\"$namespace\"}[5m])))" } + ] + }, + { + "id": 6, + "type": "timeseries", + "title": "Memory by pod (working set, top 10)", + "gridPos": { "h": 8, "w": 12, "x": 12, "y": 4 }, + "datasource": { "type": "prometheus", "uid": "${datasource}" }, + "fieldConfig": { "defaults": { "unit": "bytes" }, "overrides": [] }, + "targets": [ + { "refId": "A", "datasource": { "type": "prometheus", "uid": "${datasource}" }, + "legendFormat": "{{namespace}}/{{pod}}", + "expr": "topk(10, sum by (namespace, pod) (container_memory_working_set_bytes{container!=\"\", namespace=~\"$namespace\"}))" } + ] + }, + { + "id": 7, + "type": "table", + "title": "Deployments - available replicas", + "gridPos": { "h": 8, "w": 12, "x": 0, "y": 12 }, + "datasource": { "type": "prometheus", "uid": "${datasource}" }, + "targets": [ + { "refId": "A", "datasource": { "type": "prometheus", "uid": "${datasource}" }, + "format": "table", "instant": true, + "expr": "kube_deployment_status_replicas_available{namespace=~\"$namespace\"}" } + ] + }, + { + "id": 8, + "type": "timeseries", + "title": "Restart rate by pod (15m)", + "gridPos": { "h": 8, "w": 12, "x": 12, "y": 12 }, + "datasource": { "type": "prometheus", "uid": "${datasource}" }, + "fieldConfig": { "defaults": { "unit": "short" }, "overrides": [] }, + "targets": [ + { "refId": "A", "datasource": { "type": "prometheus", "uid": "${datasource}" }, + "legendFormat": "{{namespace}}/{{pod}}", + "expr": "sum by (namespace, pod) (rate(kube_pod_container_status_restarts_total{namespace=~\"$namespace\"}[15m]))" } + ] + } + ] +} diff --git a/infra/k8s/monitor/grafana.yaml b/infra/k8s/monitor/grafana.yaml new file mode 100644 index 0000000..5ef64e9 --- /dev/null +++ b/infra/k8s/monitor/grafana.yaml @@ -0,0 +1,133 @@ +# Grafana — Prometheus datasource'u onceden tanimli gelir (provisioning). +# Admin parolasi repoda YOK: 'grafana-admin' SealedSecret (bkz. docs/MONITORING.md). +apiVersion: v1 +kind: ConfigMap +metadata: + name: grafana-datasources + namespace: deephorizon-monitor +data: + datasource.yaml: | + apiVersion: 1 + datasources: + - name: Prometheus + type: prometheus + access: proxy + url: http://prometheus.deephorizon-monitor.svc:9090 + isDefault: true +--- +# Dashboard provider — /etc/grafana/dashboards'daki JSON'lari otomatik yukler. +apiVersion: v1 +kind: ConfigMap +metadata: + name: grafana-dashboard-provider + namespace: deephorizon-monitor +data: + provider.yaml: | + apiVersion: 1 + providers: + - name: deephorizon + orgId: 1 + type: file + disableDeletion: true + updateIntervalSeconds: 30 + allowUiUpdates: true + options: + path: /etc/grafana/dashboards +--- +apiVersion: v1 +kind: PersistentVolumeClaim +metadata: + name: grafana-data + namespace: deephorizon-monitor +spec: + accessModes: ["ReadWriteOnce"] + resources: + requests: + storage: 5Gi +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: grafana + namespace: deephorizon-monitor +spec: + replicas: 1 + strategy: + type: Recreate + selector: + matchLabels: + app: grafana + template: + metadata: + labels: + app: grafana + spec: + securityContext: + # Grafana imaji 472 kullanicisiyla kosar; PVC'yi bu gruba yazilabilir yap. + fsGroup: 472 + runAsNonRoot: true + runAsUser: 472 + containers: + - name: grafana + image: grafana/grafana:11.4.0 + env: + - name: GF_SECURITY_ADMIN_USER + value: admin + - name: GF_SECURITY_ADMIN_PASSWORD + valueFrom: + secretKeyRef: + name: grafana-admin + key: admin-password + ports: + - name: http + containerPort: 3000 + resources: + requests: + cpu: 100m + memory: 128Mi + limits: + cpu: 500m + memory: 512Mi + readinessProbe: + httpGet: + path: /api/health + port: http + initialDelaySeconds: 10 + volumeMounts: + - name: data + mountPath: /var/lib/grafana + - name: datasources + mountPath: /etc/grafana/provisioning/datasources + - name: dashboard-provider + mountPath: /etc/grafana/provisioning/dashboards + - name: dashboards + mountPath: /etc/grafana/dashboards + volumes: + - name: data + persistentVolumeClaim: + claimName: grafana-data + - name: datasources + configMap: + name: grafana-datasources + - name: dashboard-provider + configMap: + name: grafana-dashboard-provider + - name: dashboards + configMap: + name: grafana-dashboards +--- +# NodePort 30030 -> NPM (auth arkasi) veya LAN. UFW yalniz yerel subnet'e acar. +apiVersion: v1 +kind: Service +metadata: + name: grafana + namespace: deephorizon-monitor +spec: + type: NodePort + selector: + app: grafana + ports: + - name: http + port: 3000 + targetPort: http + nodePort: 30030 diff --git a/infra/k8s/monitor/kube-state-metrics.yaml b/infra/k8s/monitor/kube-state-metrics.yaml new file mode 100644 index 0000000..01575ff --- /dev/null +++ b/infra/k8s/monitor/kube-state-metrics.yaml @@ -0,0 +1,94 @@ +# kube-state-metrics — K8s obje durumlarini metrige cevirir (deployment/pod/pvc/job). +apiVersion: v1 +kind: ServiceAccount +metadata: + name: kube-state-metrics + namespace: deephorizon-monitor +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRole +metadata: + name: kube-state-metrics +rules: + - apiGroups: [""] + resources: ["configmaps", "secrets", "nodes", "pods", "services", "resourcequotas", + "replicationcontrollers", "limitranges", "persistentvolumeclaims", + "persistentvolumes", "namespaces", "endpoints"] + verbs: ["list", "watch"] + - apiGroups: ["apps"] + resources: ["statefulsets", "daemonsets", "deployments", "replicasets"] + verbs: ["list", "watch"] + - apiGroups: ["batch"] + resources: ["cronjobs", "jobs"] + verbs: ["list", "watch"] + - apiGroups: ["autoscaling"] + resources: ["horizontalpodautoscalers"] + verbs: ["list", "watch"] + - apiGroups: ["policy"] + resources: ["poddisruptionbudgets"] + verbs: ["list", "watch"] +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRoleBinding +metadata: + name: kube-state-metrics +roleRef: + apiGroup: rbac.authorization.k8s.io + kind: ClusterRole + name: kube-state-metrics +subjects: + - kind: ServiceAccount + name: kube-state-metrics + namespace: deephorizon-monitor +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: kube-state-metrics + namespace: deephorizon-monitor +spec: + replicas: 1 + selector: + matchLabels: + app: kube-state-metrics + template: + metadata: + labels: + app: kube-state-metrics + spec: + serviceAccountName: kube-state-metrics + securityContext: + runAsNonRoot: true + runAsUser: 65534 + containers: + - name: kube-state-metrics + image: registry.k8s.io/kube-state-metrics/kube-state-metrics:v2.14.0 + ports: + - name: metrics + containerPort: 8080 + resources: + requests: + cpu: 50m + memory: 128Mi + limits: + cpu: 200m + memory: 256Mi + readinessProbe: + httpGet: + path: /healthz + port: metrics + initialDelaySeconds: 5 +--- +apiVersion: v1 +kind: Service +metadata: + name: kube-state-metrics + namespace: deephorizon-monitor +spec: + type: ClusterIP + selector: + app: kube-state-metrics + ports: + - name: metrics + port: 8080 + targetPort: metrics diff --git a/infra/k8s/monitor/kustomization.yaml b/infra/k8s/monitor/kustomization.yaml new file mode 100644 index 0000000..6c88511 --- /dev/null +++ b/infra/k8s/monitor/kustomization.yaml @@ -0,0 +1,19 @@ +apiVersion: kustomize.config.k8s.io/v1beta1 +kind: Kustomization +resources: + - namespace.yaml + - prometheus.yaml + - node-exporter.yaml + - kube-state-metrics.yaml + - grafana.yaml + +# Dashboard JSON'lari 'grafana-dashboards' ConfigMap'ine cevrilir. Yeni pano = buraya dosya ekle. +configMapGenerator: + - name: grafana-dashboards + namespace: deephorizon-monitor + files: + - dashboards/deephorizon-cluster.json + +# Hash sonekini kapat: ConfigMap adi sabit kalsin (grafana.yaml ada gore baglaniyor). +generatorOptions: + disableNameSuffixHash: true diff --git a/infra/k8s/monitor/namespace.yaml b/infra/k8s/monitor/namespace.yaml new file mode 100644 index 0000000..70ee5dd --- /dev/null +++ b/infra/k8s/monitor/namespace.yaml @@ -0,0 +1,4 @@ +apiVersion: v1 +kind: Namespace +metadata: + name: deephorizon-monitor diff --git a/infra/k8s/monitor/node-exporter.yaml b/infra/k8s/monitor/node-exporter.yaml new file mode 100644 index 0000000..45e3082 --- /dev/null +++ b/infra/k8s/monitor/node-exporter.yaml @@ -0,0 +1,77 @@ +# node-exporter — node CPU/RAM/disk/ag metrikleri (DaemonSet, host /proc /sys / okur). +apiVersion: apps/v1 +kind: DaemonSet +metadata: + name: node-exporter + namespace: deephorizon-monitor +spec: + selector: + matchLabels: + app: node-exporter + template: + metadata: + labels: + app: node-exporter + spec: + hostNetwork: true + hostPID: true + # Tek node ayni zamanda control-plane; taint varsa yine schedule olsun. + tolerations: + - operator: Exists + securityContext: + runAsNonRoot: true + runAsUser: 65534 + containers: + - name: node-exporter + image: quay.io/prometheus/node-exporter:v1.8.2 + args: + - --path.procfs=/host/proc + - --path.sysfs=/host/sys + - --path.rootfs=/host/root + - --collector.filesystem.mount-points-exclude=^/(sys|proc|dev|host|etc)($|/) + ports: + - name: metrics + containerPort: 9100 + resources: + requests: + cpu: 50m + memory: 64Mi + limits: + cpu: 200m + memory: 128Mi + volumeMounts: + - name: proc + mountPath: /host/proc + readOnly: true + - name: sys + mountPath: /host/sys + readOnly: true + - name: root + mountPath: /host/root + mountPropagation: HostToContainer + readOnly: true + volumes: + - name: proc + hostPath: + path: /proc + - name: sys + hostPath: + path: /sys + - name: root + hostPath: + path: / +--- +# Prometheus node-exporter job'i bu Service'in endpoint'lerini "keep" ediyor. +apiVersion: v1 +kind: Service +metadata: + name: node-exporter + namespace: deephorizon-monitor +spec: + type: ClusterIP + selector: + app: node-exporter + ports: + - name: metrics + port: 9100 + targetPort: metrics diff --git a/infra/k8s/monitor/prometheus.yaml b/infra/k8s/monitor/prometheus.yaml new file mode 100644 index 0000000..cade7f9 --- /dev/null +++ b/infra/k8s/monitor/prometheus.yaml @@ -0,0 +1,205 @@ +# Prometheus — metrik toplayici + zaman serisi deposu. +# RBAC: cluster'daki node/pod/service/endpoint'leri kesfetmek icin salt-okuma. +apiVersion: v1 +kind: ServiceAccount +metadata: + name: prometheus + namespace: deephorizon-monitor +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRole +metadata: + name: prometheus +rules: + - apiGroups: [""] + resources: ["nodes", "nodes/proxy", "nodes/metrics", "services", "endpoints", "pods"] + verbs: ["get", "list", "watch"] + - nonResourceURLs: ["/metrics"] + verbs: ["get"] +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRoleBinding +metadata: + name: prometheus +roleRef: + apiGroup: rbac.authorization.k8s.io + kind: ClusterRole + name: prometheus +subjects: + - kind: ServiceAccount + name: prometheus + namespace: deephorizon-monitor +--- +# Scrape config — hedefleri Kubernetes service discovery ile otomatik bulur. +apiVersion: v1 +kind: ConfigMap +metadata: + name: prometheus-config + namespace: deephorizon-monitor +data: + prometheus.yml: | + global: + scrape_interval: 15s + evaluation_interval: 15s + + scrape_configs: + - job_name: prometheus + static_configs: + - targets: ["localhost:9090"] + + # Node kaynak metrikleri (kubelet uzerinden cAdvisor konteyner metrikleri). + - job_name: kubernetes-nodes-cadvisor + scheme: https + tls_config: + ca_file: /var/run/secrets/kubernetes.io/serviceaccount/ca.crt + insecure_skip_verify: true + bearer_token_file: /var/run/secrets/kubernetes.io/serviceaccount/token + kubernetes_sd_configs: + - role: node + relabel_configs: + - action: labelmap + regex: __meta_kubernetes_node_label_(.+) + - target_label: __address__ + replacement: kubernetes.default.svc:443 + - source_labels: [__meta_kubernetes_node_name] + regex: (.+) + target_label: __metrics_path__ + replacement: /api/v1/nodes/${1}/proxy/metrics/cadvisor + + # kube-state-metrics: deployment/pod/pvc durumlari (restart, hazir replika...). + - job_name: kube-state-metrics + static_configs: + - targets: ["kube-state-metrics.deephorizon-monitor.svc:8080"] + + # node-exporter: node CPU/RAM/disk/ag (Service endpoint'lerinden bulunur). + - job_name: node-exporter + kubernetes_sd_configs: + - role: endpoints + relabel_configs: + - source_labels: [__meta_kubernetes_endpoints_name] + action: keep + regex: node-exporter + + # GPU (L40S) — dcgm-exporter anotasyon tasimadigi icin Service'ini dogrudan hedefliyoruz. + - job_name: dcgm-exporter + kubernetes_sd_configs: + - role: endpoints + namespaces: + names: [gpu-operator-resources] + relabel_configs: + - source_labels: [__meta_kubernetes_endpoints_name] + action: keep + regex: nvidia-dcgm-exporter + + # Otomatik kesif: prometheus.io/scrape: "true" tasiyan her pod (api/inference). + - job_name: kubernetes-pods + kubernetes_sd_configs: + - role: pod + relabel_configs: + - source_labels: [__meta_kubernetes_pod_annotation_prometheus_io_scrape] + action: keep + regex: "true" + - source_labels: [__meta_kubernetes_pod_annotation_prometheus_io_path] + action: replace + target_label: __metrics_path__ + regex: (.+) + - source_labels: [__address__, __meta_kubernetes_pod_annotation_prometheus_io_port] + action: replace + regex: ([^:]+)(?::\d+)?;(\d+) + replacement: $1:$2 + target_label: __address__ + - source_labels: [__meta_kubernetes_namespace] + target_label: namespace + - source_labels: [__meta_kubernetes_pod_name] + target_label: pod +--- +# TSDB icin kalici disk. Silinirse metrik gecmisi gider (hostpath, tek node). +apiVersion: v1 +kind: PersistentVolumeClaim +metadata: + name: prometheus-data + namespace: deephorizon-monitor +spec: + accessModes: ["ReadWriteOnce"] + resources: + requests: + storage: 20Gi +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: prometheus + namespace: deephorizon-monitor +spec: + replicas: 1 + # PVC ReadWriteOnce; guncellemede eski pod diski birakmadan yeni pod baglanamaz. + strategy: + type: Recreate + selector: + matchLabels: + app: prometheus + template: + metadata: + labels: + app: prometheus + spec: + serviceAccountName: prometheus + securityContext: + # Resmi imaj nobody (65534) ile kosar; PVC'yi bu gruba yazilabilir yap. + fsGroup: 65534 + runAsNonRoot: true + runAsUser: 65534 + containers: + - name: prometheus + image: prom/prometheus:v3.1.0 + args: + - --config.file=/etc/prometheus/prometheus.yml + - --storage.tsdb.path=/prometheus + - --storage.tsdb.retention.time=15d + ports: + - name: web + containerPort: 9090 + resources: + requests: + cpu: 250m + memory: 512Mi + limits: + cpu: "1" + memory: 2Gi + readinessProbe: + httpGet: + path: /-/ready + port: web + initialDelaySeconds: 10 + livenessProbe: + httpGet: + path: /-/healthy + port: web + initialDelaySeconds: 30 + volumeMounts: + - name: config + mountPath: /etc/prometheus + - name: data + mountPath: /prometheus + volumes: + - name: config + configMap: + name: prometheus-config + - name: data + persistentVolumeClaim: + claimName: prometheus-data +--- +# Cluster ici. Grafana ve dahili scrape bu adrese konusur. Disari ACILMAZ. +apiVersion: v1 +kind: Service +metadata: + name: prometheus + namespace: deephorizon-monitor +spec: + type: ClusterIP + selector: + app: prometheus + ports: + - name: web + port: 9090 + targetPort: web