diff --git a/add-on/blackbox-exporter.md b/add-on/monitoring/blackbox-exporter.md similarity index 74% rename from add-on/blackbox-exporter.md rename to add-on/monitoring/blackbox-exporter.md index ff3b867..42eab03 100644 --- a/add-on/blackbox-exporter.md +++ b/add-on/monitoring/blackbox-exporter.md @@ -359,7 +359,224 @@ Interpretazione: - probe_duration_seconds: durata della probe. -8. CREA ALERT PROMETHEUSRULE +8. CREA DASHBOARD GRAFANA PER LE PROBE +====================================== + +Obiettivo: +- visualizzare lo stato delle probe HTTP; +- vedere quali target sono UP o DOWN; +- controllare status code e latenza per ogni endpoint; +- filtrare la dashboard per job e target. + +COMANDO - apri Grafana in locale con port-forward +```bash +kubectl port-forward -n monitoring svc/kube-prometheus-stack-grafana 3000:80 +``` + +Poi apri nel browser: + +```text +http://localhost:3000 +``` + +COMANDO - recupera password admin Grafana se non la conosci +```bash +kubectl get secret -n monitoring kube-prometheus-stack-grafana \ + -o jsonpath="{.data.admin-password}" | base64 -d +``` + +Credenziali predefinite tipiche: +```text +utente: admin +password: valore recuperato dal secret +``` + +NOTA - datasource Prometheus +Con kube-prometheus-stack il datasource Prometheus di solito e' gia' configurato in Grafana. +Verifica da Grafana: + +```text +Connections -> Data sources -> Prometheus +``` + +CREAZIONE DASHBOARD DA INTERFACCIA + +1. Vai su Dashboards -> New -> New dashboard. +2. Clicca Add visualization. +3. Seleziona il datasource Prometheus. +4. Crea i pannelli usando le query sotto. +5. Salva la dashboard con nome, ad esempio: + +```text +Blackbox HTTP Probes +``` + +VARIABILI CONSIGLIATE + +Variabile job: +```text +Name: job +Type: Query +Data source: Prometheus +Query: label_values(probe_success, job) +Multi-value: enabled +Include All option: enabled +``` + +Variabile instance: +```text +Name: instance +Type: Query +Data source: Prometheus +Query: label_values(probe_success{job=~"$job"}, instance) +Multi-value: enabled +Include All option: enabled +``` + +PANNELLO - stato generale probe + +Tipo pannello: Stat + +Query: +```promql +min(probe_success{job=~"$job", instance=~"$instance"}) +``` + +Configurazione consigliata: +```text +Unit: none +Thresholds: + 0 = red + 1 = green +Value mappings: + 0 -> DOWN + 1 -> UP +``` + +PANNELLO - stato per target + +Tipo pannello: State timeline oppure Table + +Query: +```promql +probe_success{job=~"$job", instance=~"$instance"} +``` + +Configurazione consigliata: +```text +Legend: {{ instance }} +Value mappings: + 0 -> DOWN + 1 -> UP +``` + +PANNELLO - target attualmente falliti + +Tipo pannello: Table + +Query: +```promql +probe_success{job=~"$job", instance=~"$instance"} == 0 +``` + +Configurazione consigliata: +```text +Legend: {{ instance }} +Mostra colonne: instance, job, value +``` + +PANNELLO - durata probe per target + +Tipo pannello: Time series + +Query: +```promql +probe_duration_seconds{job=~"$job", instance=~"$instance"} +``` + +Configurazione consigliata: +```text +Unit: seconds +Legend: {{ instance }} +Thresholds: + 1 = yellow + 2 = red +``` + +PANNELLO - status code HTTP + +Tipo pannello: Time series oppure Table + +Query: +```promql +probe_http_status_code{job=~"$job", instance=~"$instance"} +``` + +Configurazione consigliata: +```text +Unit: none +Legend: {{ instance }} +``` + +PANNELLO - percentuale disponibilita' per target + +Tipo pannello: Bar gauge oppure Table + +Query: +```promql +avg_over_time(probe_success{job=~"$job", instance=~"$instance"}[24h]) * 100 +``` + +Configurazione consigliata: +```text +Unit: percent +Min: 0 +Max: 100 +Legend: {{ instance }} +Thresholds: + 95 = yellow + 99 = green +``` + +PANNELLO - durata media nelle ultime 24 ore + +Tipo pannello: Bar gauge oppure Table + +Query: +```promql +avg_over_time(probe_duration_seconds{job=~"$job", instance=~"$instance"}[24h]) +``` + +Configurazione consigliata: +```text +Unit: seconds +Legend: {{ instance }} +``` + +QUERY RAPIDE SENZA VARIABILI + +Se vuoi creare una dashboard solo per la Probe di esempio dei servizi interni: + +```promql +probe_success{job="http-services-probe"} +probe_http_status_code{job="http-services-probe"} +probe_duration_seconds{job="http-services-probe"} +probe_success{job="http-services-probe"} == 0 +avg_over_time(probe_success{job="http-services-probe"}[24h]) * 100 +``` + +Per la Probe di esempio degli Ingress o URL esterni: + +```promql +probe_success{job="ingress-http-probe"} +probe_http_status_code{job="ingress-http-probe"} +probe_duration_seconds{job="ingress-http-probe"} +probe_success{job="ingress-http-probe"} == 0 +avg_over_time(probe_success{job="ingress-http-probe"}[24h]) * 100 +``` + + +9. CREA ALERT PROMETHEUSRULE ============================ FILE - blackbox-http-alerts.yaml @@ -414,7 +631,7 @@ kubectl describe prometheusrule blackbox-http-alerts -n monitoring ``` -9. TEST MANUALE BLACKBOX EXPORTER +10. TEST MANUALE BLACKBOX EXPORTER ================================= COMANDO - port-forward blackbox-exporter @@ -434,7 +651,7 @@ probe_http_status_code 200 ``` -10. TROUBLESHOOTING +11. TROUBLESHOOTING =================== CASO - Prometheus non vede la Probe @@ -483,7 +700,7 @@ Azioni consigliate: - aggiungere un modulo dedicato in blackbox-values.yaml se serve una configurazione HTTP specifica. -11. ORDINE DI ESECUZIONE CONSIGLIATO +12. ORDINE DI ESECUZIONE CONSIGLIATO ==================================== COMANDO - sequenza completa @@ -510,7 +727,7 @@ blackbox-http-alerts.yaml ``` -12. NOTE PER DEV, QA E PROD +13. NOTE PER DEV, QA E PROD =========================== Per separare gli ambienti e semplificare dashboard e alert, usa Probe distinte: diff --git a/add-on/monitoring/files/probe-dash-external.yaml b/add-on/monitoring/files/probe-dash-external.yaml new file mode 100644 index 0000000..ee9c9a9 --- /dev/null +++ b/add-on/monitoring/files/probe-dash-external.yaml @@ -0,0 +1,553 @@ +apiVersion: monitoring.coreos.com/v1 +kind: Probe +metadata: + name: ---probe + namespace: monitoring + labels: + release: kube-prometheus-stack +spec: + jobName: ---probe + interval: 30s + scrapeTimeout: 15s + module: http_2xx + prober: + url: blackbox-exporter.monitoring.svc.cluster.local:9115 + scheme: http + path: /probe + targets: + staticConfig: + static: + - https:// +--- +apiVersion: v1 +kind: ConfigMap +metadata: + name: ---dashboard + namespace: monitoring + labels: + grafana_dashboard: "1" +data: + ---dashboard.json: | + { + "annotations": { + "list": [] + }, + "editable": true, + "fiscalYearStartMonth": 0, + "graphTooltip": 1, + "links": [], + "panels": [ + { + "type": "stat", + "title": "Services UP", + "id": 1, + "gridPos": { + "h": 5, + "w": 6, + "x": 0, + "y": 0 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "sum(probe_success{job=\"---probe\",instance=~\"$instance\"})" + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "red", + "value": null + }, + { + "color": "green", + "value": 1 + } + ] + } + } + }, + "options": { + "reduceOptions": { + "calcs": [ + "lastNotNull" + ], + "fields": "", + "values": false + }, + "orientation": "auto", + "textMode": "auto", + "colorMode": "value", + "graphMode": "none" + } + }, + { + "type": "stat", + "title": "Services DOWN", + "id": 2, + "gridPos": { + "h": 5, + "w": 6, + "x": 6, + "y": 0 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "count(probe_success{job=\"---probe\",instance=~\"$instance\"}) - sum(probe_success{job=\"---probe\",instance=~\"$instance\"})" + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + }, + { + "color": "red", + "value": 1 + } + ] + } + } + }, + "options": { + "reduceOptions": { + "calcs": [ + "lastNotNull" + ], + "fields": "", + "values": false + }, + "orientation": "auto", + "textMode": "auto", + "colorMode": "value", + "graphMode": "none" + } + }, + { + "type": "stat", + "title": "Availability 24h", + "id": 3, + "gridPos": { + "h": 5, + "w": 6, + "x": 12, + "y": 0 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "avg(avg_over_time(probe_success{job=\"---probe\",instance=~\"$instance\"}[24h])) * 100" + } + ], + "fieldConfig": { + "defaults": { + "unit": "percent", + "min": 0, + "max": 100, + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "red", + "value": null + }, + { + "color": "orange", + "value": 99 + }, + { + "color": "green", + "value": 99.9 + } + ] + } + } + }, + "options": { + "reduceOptions": { + "calcs": [ + "lastNotNull" + ], + "fields": "", + "values": false + }, + "orientation": "auto", + "textMode": "auto", + "colorMode": "value", + "graphMode": "area" + } + }, + { + "type": "stat", + "title": "Avg Latency", + "id": 4, + "gridPos": { + "h": 5, + "w": 6, + "x": 18, + "y": 0 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "avg(probe_duration_seconds{job=\"---probe\",instance=~\"$instance\"}) * 1000" + } + ], + "fieldConfig": { + "defaults": { + "unit": "ms", + "decimals": 0 + } + }, + "options": { + "reduceOptions": { + "calcs": [ + "lastNotNull" + ], + "fields": "", + "values": false + }, + "orientation": "auto", + "textMode": "auto", + "colorMode": "value", + "graphMode": "none" + } + }, + { + "type": "table", + "title": "HTTP Services Status", + "id": 5, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 5 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "probe_success{job=\"---probe\",instance=~\"$instance\"}", + "instant": true, + "format": "table" + } + ], + "fieldConfig": { + "defaults": { + "custom": { + "align": "auto", + "cellOptions": { + "type": "color-text" + } + } + } + }, + "transformations": [ + { + "id": "organize", + "options": { + "excludeByName": { + "Time": true + }, + "renameByName": { + "Value": "Status", + "instance": "Service" + } + } + } + ], + "options": { + "showHeader": true, + "cellHeight": "sm" + } + }, + { + "type": "table", + "title": "HTTP Status Code", + "id": 6, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 5 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "probe_http_status_code{job=\"---probe\",instance=~\"$instance\"}", + "instant": true, + "format": "table" + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "align": "auto", + "cellOptions": { + "type": "color-text" + } + } + } + }, + "transformations": [ + { + "id": "organize", + "options": { + "excludeByName": { + "Time": true + }, + "renameByName": { + "Value": "HTTP Status", + "instance": "Service" + } + } + } + ], + "options": { + "showHeader": true, + "cellHeight": "sm" + } + }, + { + "type": "timeseries", + "title": "HTTP Response Time", + "id": 7, + "gridPos": { + "h": 9, + "w": 24, + "x": 0, + "y": 13 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "probe_duration_seconds{job=\"---probe\",instance=~\"$instance\"} * 1000", + "legendFormat": "{{instance}}" + } + ], + "fieldConfig": { + "defaults": { + "unit": "ms", + "decimals": 0 + } + }, + "options": { + "legend": { + "displayMode": "list", + "placement": "bottom" + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + } + }, + { + "type": "table", + "title": "Currently DOWN", + "id": 8, + "gridPos": { + "h": 7, + "w": 12, + "x": 0, + "y": 22 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "probe_success{job=\"---probe\",instance=~\"$instance\"} == 0", + "instant": true, + "format": "table" + } + ], + "fieldConfig": { + "defaults": { + "custom": { + "cellOptions": { + "type": "color-text" + } + } + } + }, + "transformations": [ + { + "id": "organize", + "options": { + "excludeByName": { + "Time": true + }, + "renameByName": { + "Value": "Status", + "instance": "Service" + } + } + } + ], + "options": { + "showHeader": true, + "cellHeight": "sm" + } + }, + { + "type": "timeseries", + "title": "Availability 24h", + "id": 9, + "gridPos": { + "h": 9, + "w": 12, + "x": 12, + "y": 22 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "avg_over_time(probe_success{job=\"---probe\",instance=~\"$instance\"}[24h]) * 100", + "legendFormat": "{{instance}}" + } + ], + "fieldConfig": { + "defaults": { + "unit": "percent", + "min": 0, + "max": 100, + "decimals": 2 + } + }, + "options": { + "legend": { + "displayMode": "list", + "placement": "bottom" + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + } + }, + { + "type": "timeseries", + "title": "P95 Latency - 1h", + "id": 10, + "gridPos": { + "h": 9, + "w": 24, + "x": 0, + "y": 31 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "quantile_over_time(0.95, probe_duration_seconds{job=\"---probe\",instance=~\"$instance\"}[1h]) * 1000", + "legendFormat": "{{instance}}" + } + ], + "fieldConfig": { + "defaults": { + "unit": "ms", + "decimals": 0 + } + }, + "options": { + "legend": { + "displayMode": "list", + "placement": "bottom" + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + } + } + ], + "refresh": "30s", + "schemaVersion": 41, + "tags": [ + "blackbox", + "http", + "monitoring" + ], + + "templating": { + "list": [ + { + "name": "instance", + "label": "Service", + "type": "query", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "definition": "label_values(probe_success{job=\"---probe\"}, instance)", + "query": { + "query": "label_values(probe_success{job=\"---probe\"}, instance)", + "refId": "StandardVariableQuery" + }, + "includeAll": true, + "multi": true, + "allValue": ".*", + "refresh": 1, + "sort": 1 + } + ] + }, + "time": { + "from": "now-24h", + "to": "now" + }, + "timepicker": {}, + "timezone": "browser", + "title": "---dashboard", + "uid": "---dashboard", + "version": 1, + "weekStart": "" + } diff --git a/add-on/monitoring/files/probe-dash-internal.yaml b/add-on/monitoring/files/probe-dash-internal.yaml new file mode 100644 index 0000000..6b5a977 --- /dev/null +++ b/add-on/monitoring/files/probe-dash-internal.yaml @@ -0,0 +1,553 @@ +apiVersion: monitoring.coreos.com/v1 +kind: Probe +metadata: + name: + namespace: monitoring + labels: + release: kube-prometheus-stack +spec: + jobName: + interval: 30s + scrapeTimeout: 15s + module: http_k8s_health + prober: + url: blackbox-exporter.monitoring.svc.cluster.local:9115 + scheme: http + path: /probe + targets: + staticConfig: + static: + - http://..svc.cluster.local:/health +--- +apiVersion: v1 +kind: ConfigMap +metadata: + name: + namespace: monitoring + labels: + grafana_dashboard: "1" +data: + .json: | + { + "annotations": { + "list": [] + }, + "editable": true, + "fiscalYearStartMonth": 0, + "graphTooltip": 1, + "links": [], + "panels": [ + { + "type": "stat", + "title": "Services UP", + "id": 1, + "gridPos": { + "h": 5, + "w": 6, + "x": 0, + "y": 0 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "sum(probe_success{job=\"\",instance=~\"$instance\"})" + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "red", + "value": null + }, + { + "color": "green", + "value": 1 + } + ] + } + } + }, + "options": { + "reduceOptions": { + "calcs": [ + "lastNotNull" + ], + "fields": "", + "values": false + }, + "orientation": "auto", + "textMode": "auto", + "colorMode": "value", + "graphMode": "none" + } + }, + { + "type": "stat", + "title": "Services DOWN", + "id": 2, + "gridPos": { + "h": 5, + "w": 6, + "x": 6, + "y": 0 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "count(probe_success{job=\"\",instance=~\"$instance\"}) - sum(probe_success{job=\"\",instance=~\"$instance\"})" + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + }, + { + "color": "red", + "value": 1 + } + ] + } + } + }, + "options": { + "reduceOptions": { + "calcs": [ + "lastNotNull" + ], + "fields": "", + "values": false + }, + "orientation": "auto", + "textMode": "auto", + "colorMode": "value", + "graphMode": "none" + } + }, + { + "type": "stat", + "title": "Availability 24h", + "id": 3, + "gridPos": { + "h": 5, + "w": 6, + "x": 12, + "y": 0 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "avg(avg_over_time(probe_success{job=\"\",instance=~\"$instance\"}[24h])) * 100" + } + ], + "fieldConfig": { + "defaults": { + "unit": "percent", + "min": 0, + "max": 100, + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "red", + "value": null + }, + { + "color": "orange", + "value": 99 + }, + { + "color": "green", + "value": 99.9 + } + ] + } + } + }, + "options": { + "reduceOptions": { + "calcs": [ + "lastNotNull" + ], + "fields": "", + "values": false + }, + "orientation": "auto", + "textMode": "auto", + "colorMode": "value", + "graphMode": "area" + } + }, + { + "type": "stat", + "title": "Avg Latency", + "id": 4, + "gridPos": { + "h": 5, + "w": 6, + "x": 18, + "y": 0 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "avg(probe_duration_seconds{job=\"\",instance=~\"$instance\"}) * 1000" + } + ], + "fieldConfig": { + "defaults": { + "unit": "ms", + "decimals": 0 + } + }, + "options": { + "reduceOptions": { + "calcs": [ + "lastNotNull" + ], + "fields": "", + "values": false + }, + "orientation": "auto", + "textMode": "auto", + "colorMode": "value", + "graphMode": "none" + } + }, + { + "type": "table", + "title": "HTTP Services Status", + "id": 5, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 5 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "probe_success{job=\"\",instance=~\"$instance\"}", + "instant": true, + "format": "table" + } + ], + "fieldConfig": { + "defaults": { + "custom": { + "align": "auto", + "cellOptions": { + "type": "color-text" + } + } + } + }, + "transformations": [ + { + "id": "organize", + "options": { + "excludeByName": { + "Time": true + }, + "renameByName": { + "Value": "Status", + "instance": "Service" + } + } + } + ], + "options": { + "showHeader": true, + "cellHeight": "sm" + } + }, + { + "type": "table", + "title": "HTTP Status Code", + "id": 6, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 5 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "probe_http_status_code{job=\"\",instance=~\"$instance\"}", + "instant": true, + "format": "table" + } + ], + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "align": "auto", + "cellOptions": { + "type": "color-text" + } + } + } + }, + "transformations": [ + { + "id": "organize", + "options": { + "excludeByName": { + "Time": true + }, + "renameByName": { + "Value": "HTTP Status", + "instance": "Service" + } + } + } + ], + "options": { + "showHeader": true, + "cellHeight": "sm" + } + }, + { + "type": "timeseries", + "title": "HTTP Response Time", + "id": 7, + "gridPos": { + "h": 9, + "w": 24, + "x": 0, + "y": 13 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "probe_duration_seconds{job=\"\",instance=~\"$instance\"} * 1000", + "legendFormat": "{{instance}}" + } + ], + "fieldConfig": { + "defaults": { + "unit": "ms", + "decimals": 0 + } + }, + "options": { + "legend": { + "displayMode": "list", + "placement": "bottom" + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + } + }, + { + "type": "table", + "title": "Currently DOWN", + "id": 8, + "gridPos": { + "h": 7, + "w": 12, + "x": 0, + "y": 22 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "probe_success{job=\"\",instance=~\"$instance\"} == 0", + "instant": true, + "format": "table" + } + ], + "fieldConfig": { + "defaults": { + "custom": { + "cellOptions": { + "type": "color-text" + } + } + } + }, + "transformations": [ + { + "id": "organize", + "options": { + "excludeByName": { + "Time": true + }, + "renameByName": { + "Value": "Status", + "instance": "Service" + } + } + } + ], + "options": { + "showHeader": true, + "cellHeight": "sm" + } + }, + { + "type": "timeseries", + "title": "Availability 24h", + "id": 9, + "gridPos": { + "h": 9, + "w": 12, + "x": 12, + "y": 22 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "avg_over_time(probe_success{job=\"\",instance=~\"$instance\"}[24h]) * 100", + "legendFormat": "{{instance}}" + } + ], + "fieldConfig": { + "defaults": { + "unit": "percent", + "min": 0, + "max": 100, + "decimals": 2 + } + }, + "options": { + "legend": { + "displayMode": "list", + "placement": "bottom" + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + } + }, + { + "type": "timeseries", + "title": "P95 Latency - 1h", + "id": 10, + "gridPos": { + "h": 9, + "w": 24, + "x": 0, + "y": 31 + }, + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "refId": "A", + "expr": "quantile_over_time(0.95, probe_duration_seconds{job=\"\",instance=~\"$instance\"}[1h]) * 1000", + "legendFormat": "{{instance}}" + } + ], + "fieldConfig": { + "defaults": { + "unit": "ms", + "decimals": 0 + } + }, + "options": { + "legend": { + "displayMode": "list", + "placement": "bottom" + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + } + } + ], + "refresh": "30s", + "schemaVersion": 41, + "tags": [ + "blackbox", + "http", + "monitoring" + ], + + "templating": { + "list": [ + { + "name": "instance", + "label": "Service", + "type": "query", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "definition": "label_values(probe_success{job=\"\"}, instance)", + "query": { + "query": "label_values(probe_success{job=\"\"}, instance)", + "refId": "StandardVariableQuery" + }, + "includeAll": true, + "multi": true, + "allValue": ".*", + "refresh": 1, + "sort": 1 + } + ] + }, + "time": { + "from": "now-24h", + "to": "now" + }, + "timepicker": {}, + "timezone": "browser", + "title": "", + "uid": "", + "version": 1, + "weekStart": "" + } diff --git a/pipeline/customize.sh b/pipeline/customize.sh index 50c6b8e..35cb2a1 100644 --- a/pipeline/customize.sh +++ b/pipeline/customize.sh @@ -6,6 +6,19 @@ VALUES_DIR="env/$ENV" PROPERTIES_FILE="properties.env" YAML_DIR="kubernetes" +# Ricava il nome progetto dalle variabili esposte da Gitea/GitHub Actions. +# Fallback locale: nome della directory del repository. +PROJECT_NAME="${GITHUB_REPOSITORY_NAME:-${GITEA_REPOSITORY_NAME:-}}" +if [ -z "$PROJECT_NAME" ] && [ -n "${GITHUB_REPOSITORY:-}" ]; then + PROJECT_NAME="${GITHUB_REPOSITORY##*/}" +fi +if [ -z "$PROJECT_NAME" ] && [ -n "${GITEA_REPOSITORY:-}" ]; then + PROJECT_NAME="${GITEA_REPOSITORY##*/}" +fi +if [ -z "$PROJECT_NAME" ]; then + PROJECT_NAME=$(basename "$(git rev-parse --show-toplevel 2>/dev/null || pwd)") +fi + # Estrai l'hash completo del commit e crea una variabile temporanea per la sostituzione TAG=$(git rev-parse HEAD) TMP_TAG_FILE=$(mktemp) @@ -24,6 +37,9 @@ if [ ! -f "$PROPERTIES_FILE" ]; then exit 1 fi +# Costruisce il nome del namespace come - +NAMESPACE="${PROJECT_NAME}-${ENV}" + # Crea una lista key=value temporanea partendo da tutti i file *.env in env/$ENV e aggiunge env dinamico > "$TMP_VALUES_FILE" for env_file in "$VALUES_DIR"/*.env; do @@ -31,6 +47,15 @@ for env_file in "$VALUES_DIR"/*.env; do printf '\n' >> "$TMP_VALUES_FILE" done printf '\nenv=%s\n' "$ENV" >> "$TMP_VALUES_FILE" +printf 'namespace=%s\n' "$NAMESPACE" >> "$TMP_VALUES_FILE" +printf 'project=%s\n' "$PROJECT_NAME" >> "$TMP_VALUES_FILE" + +# Ricava endpoint se presente e genera endpoint-nodot sostituendo '.' con '-' +ENDPOINT_VAL=$(grep -E '^[[:space:]]*endpoint[[:space:]]*=' "$TMP_VALUES_FILE" "$PROPERTIES_FILE" 2>/dev/null | tail -n 1 | sed -E 's/.*endpoint[[:space:]]*=[[:space:]]*//' | tr -d '\r\n[:space:]"' || true) +if [ -n "$ENDPOINT_VAL" ]; then + ENDPOINT_NODOT=$(echo "$ENDPOINT_VAL" | tr '.' '-') + printf 'endpoint-nodot=%s\n' "$ENDPOINT_NODOT" >> "$TMP_VALUES_FILE" +fi # Trova tutti i file .yaml nella directory kubernetes e sottodirectory find "$YAML_DIR" -type f -name "*.yaml" | while read YAML_FILE; do diff --git a/pipeline/deploy.sh b/pipeline/deploy.sh index e4d1dfb..1ea3c62 100644 --- a/pipeline/deploy.sh +++ b/pipeline/deploy.sh @@ -51,8 +51,13 @@ fi # --------------------------------------------------------------------------- find "$YAML_DIR" -type d | while read DIR; do if ls "$DIR"/*.yaml 1> /dev/null 2>&1; then - echo "Deploy delle risorse nella directory $DIR..." - kubectl --kubeconfig=./kubeconfig apply -f "$DIR" + if [ "$(basename "$DIR")" = "monitoring" ]; then + echo "Deploy delle risorse di monitoring nella directory $DIR con kubeconfig dedicato..." + kubectl --kubeconfig=/root/work/pipeline/kc apply -f "$DIR" + else + echo "Deploy delle risorse nella directory $DIR..." + kubectl --kubeconfig=./kubeconfig apply -f "$DIR" + fi fi done