varie
This commit is contained in:
@@ -0,0 +1,754 @@
|
||||
BLACKBOX EXPORTER - GUIDA OPERATIVA KUBERNETES
|
||||
==============================================
|
||||
|
||||
Stack di partenza
|
||||
-----------------
|
||||
Nel cluster e' gia' installato kube-prometheus-stack nel namespace monitoring:
|
||||
|
||||
```text
|
||||
helm list -A | grep prometheus
|
||||
|
||||
kube-prometheus-stack monitoring 4 2026-08-08 23:55:14.193903052 +0000 UTC deployed kube-prometheus-stack-88.1.5
|
||||
```
|
||||
|
||||
Obiettivo
|
||||
---------
|
||||
Installare prometheus-blackbox-exporter per eseguire test HTTP/HTTPS su servizi Kubernetes interni, Ingress o URL esterni, esportando le metriche verso Prometheus tramite Prometheus Operator.
|
||||
|
||||
Metriche principali:
|
||||
|
||||
```text
|
||||
probe_success
|
||||
probe_duration_seconds
|
||||
probe_http_status_code
|
||||
probe_http_ssl
|
||||
probe_dns_lookup_time_seconds
|
||||
```
|
||||
|
||||
Legenda
|
||||
-------
|
||||
- COMANDO: istruzione da eseguire nel terminale.
|
||||
- FILE: file da creare o aggiornare.
|
||||
- NOTA: informazione da verificare prima di procedere.
|
||||
- OUTPUT ATTESO: risultato indicativo del comando.
|
||||
|
||||
Prerequisiti
|
||||
------------
|
||||
- Accesso kubectl al cluster.
|
||||
- Helm installato.
|
||||
- Namespace monitoring presente o creabile.
|
||||
- kube-prometheus-stack installato con Prometheus Operator.
|
||||
- CRD Probe disponibile: probes.monitoring.coreos.com.
|
||||
|
||||
|
||||
1. VERIFICA PROMETHEUS E PROMETHEUS OPERATOR
|
||||
============================================
|
||||
|
||||
COMANDO - verifica release Helm Prometheus
|
||||
```bash
|
||||
helm list -A | grep prometheus
|
||||
```
|
||||
|
||||
COMANDO - verifica pod Prometheus
|
||||
```bash
|
||||
kubectl get pods -A | grep prometheus
|
||||
```
|
||||
|
||||
COMANDO - verifica CRD Probe
|
||||
```bash
|
||||
kubectl get crd probes.monitoring.coreos.com
|
||||
```
|
||||
|
||||
COMANDO - verifica label delle risorse Prometheus
|
||||
```bash
|
||||
kubectl get prometheus -A --show-labels
|
||||
```
|
||||
|
||||
COMANDO - controlla selector della risorsa Prometheus
|
||||
```bash
|
||||
kubectl get prometheus -n monitoring -o yaml
|
||||
```
|
||||
|
||||
NOTA - label dei manifest
|
||||
Dal comando `kubectl get prometheus -A --show-labels` risulta che Prometheus seleziona le risorse con la label `release=kube-prometheus-stack`.
|
||||
Negli esempi sotto viene quindi usata questa label:
|
||||
|
||||
```yaml
|
||||
release: kube-prometheus-stack
|
||||
```
|
||||
|
||||
Se in futuro il tuo Prometheus usa un selector diverso, sostituisci questa label nei manifest Probe e PrometheusRule.
|
||||
Controlla soprattutto questi campi nella risorsa Prometheus:
|
||||
|
||||
```yaml
|
||||
spec:
|
||||
probeSelector:
|
||||
probeNamespaceSelector:
|
||||
ruleSelector:
|
||||
ruleNamespaceSelector:
|
||||
```
|
||||
|
||||
|
||||
2. AGGIUNTA REPOSITORY HELM
|
||||
===========================
|
||||
|
||||
COMANDO - aggiungi repository prometheus-community
|
||||
```bash
|
||||
helm repo add prometheus-community https://prometheus-community.github.io/helm-charts
|
||||
```
|
||||
|
||||
COMANDO - aggiorna indice chart Helm
|
||||
```bash
|
||||
helm repo update
|
||||
```
|
||||
|
||||
|
||||
3. CREA FILE VALUES HELM
|
||||
========================
|
||||
|
||||
FILE - blackbox-values.yaml
|
||||
|
||||
Descrizione:
|
||||
- configura i moduli HTTP usati da blackbox-exporter;
|
||||
- espone blackbox-exporter come Service ClusterIP sulla porta 9115;
|
||||
- lascia disabilitato ServiceMonitor perche' lo scraping dei target viene configurato tramite Probe.
|
||||
|
||||
COMANDO - crea blackbox-values.yaml
|
||||
```bash
|
||||
cat > blackbox-values.yaml <<'EOF'
|
||||
fullnameOverride: blackbox-exporter
|
||||
|
||||
config:
|
||||
modules:
|
||||
http_2xx:
|
||||
prober: http
|
||||
timeout: 10s
|
||||
http:
|
||||
method: GET
|
||||
preferred_ip_protocol: ip4
|
||||
valid_http_versions:
|
||||
- HTTP/1.1
|
||||
- HTTP/2.0
|
||||
valid_status_codes:
|
||||
- 200
|
||||
- 204
|
||||
- 301
|
||||
- 302
|
||||
|
||||
http_post_2xx:
|
||||
prober: http
|
||||
timeout: 10s
|
||||
http:
|
||||
method: POST
|
||||
preferred_ip_protocol: ip4
|
||||
valid_status_codes:
|
||||
- 200
|
||||
- 201
|
||||
- 202
|
||||
- 204
|
||||
|
||||
http_k8s_health:
|
||||
prober: http
|
||||
timeout: 10s
|
||||
http:
|
||||
method: GET
|
||||
preferred_ip_protocol: ip4
|
||||
fail_if_ssl: false
|
||||
fail_if_not_ssl: false
|
||||
valid_status_codes:
|
||||
- 200
|
||||
|
||||
service:
|
||||
type: ClusterIP
|
||||
port: 9115
|
||||
|
||||
serviceMonitor:
|
||||
enabled: false
|
||||
|
||||
resources:
|
||||
requests:
|
||||
cpu: 50m
|
||||
memory: 64Mi
|
||||
limits:
|
||||
cpu: 200m
|
||||
memory: 256Mi
|
||||
EOF
|
||||
```
|
||||
|
||||
|
||||
4. INSTALLA BLACKBOX EXPORTER
|
||||
=============================
|
||||
|
||||
COMANDO - installa o aggiorna blackbox-exporter
|
||||
```bash
|
||||
helm upgrade --install blackbox-exporter prometheus-community/prometheus-blackbox-exporter \
|
||||
--namespace monitoring \
|
||||
--create-namespace \
|
||||
-f blackbox-values.yaml
|
||||
```
|
||||
|
||||
COMANDO - verifica pod blackbox-exporter
|
||||
```bash
|
||||
kubectl get pods -n monitoring | grep blackbox
|
||||
```
|
||||
|
||||
COMANDO - verifica Service blackbox-exporter
|
||||
```bash
|
||||
kubectl get svc -n monitoring | grep blackbox
|
||||
```
|
||||
|
||||
OUTPUT ATTESO
|
||||
```text
|
||||
blackbox-exporter ClusterIP <cluster-ip> <none> 9115/TCP
|
||||
```
|
||||
|
||||
|
||||
5. CREA PROBE HTTP PER SERVIZI KUBERNETES INTERNI
|
||||
=================================================
|
||||
|
||||
FILE - blackbox-probes.yaml
|
||||
|
||||
Descrizione:
|
||||
- testa endpoint HTTP interni tramite DNS Kubernetes;
|
||||
- usa il modulo http_k8s_health definito in blackbox-values.yaml;
|
||||
- invia le metriche a Prometheus tramite la risorsa Probe del Prometheus Operator.
|
||||
|
||||
Da personalizzare:
|
||||
- my-service
|
||||
- my-namespace
|
||||
- porta del servizio
|
||||
- path HTTP, ad esempio /health, /ready o /
|
||||
- label release se diversa da prometheus
|
||||
|
||||
COMANDO - crea blackbox-probes.yaml
|
||||
```bash
|
||||
cat > blackbox-probes.yaml <<'EOF'
|
||||
apiVersion: monitoring.coreos.com/v1
|
||||
kind: Probe
|
||||
metadata:
|
||||
name: http-services-probe
|
||||
namespace: monitoring
|
||||
labels:
|
||||
release: kube-prometheus-stack
|
||||
spec:
|
||||
jobName: http-services-probe
|
||||
interval: 30s
|
||||
scrapeTimeout: 15s
|
||||
module: http_k8s_health
|
||||
prober:
|
||||
url: blackbox-exporter.monitoring.svc.cluster.local:9115
|
||||
scheme: http
|
||||
path: /probe
|
||||
targets:
|
||||
staticConfig:
|
||||
static:
|
||||
- http://my-service.my-namespace.svc.cluster.local:8080/health
|
||||
- http://another-service.default.svc.cluster.local:80/
|
||||
EOF
|
||||
```
|
||||
|
||||
COMANDO - applica Probe servizi interni
|
||||
```bash
|
||||
kubectl apply -f blackbox-probes.yaml
|
||||
```
|
||||
|
||||
COMANDO - verifica Probe
|
||||
```bash
|
||||
kubectl get probe -n monitoring
|
||||
kubectl describe probe http-services-probe -n monitoring
|
||||
```
|
||||
|
||||
NOTA - campi importanti della Probe
|
||||
```yaml
|
||||
spec:
|
||||
module: http_k8s_health
|
||||
prober:
|
||||
url: blackbox-exporter.monitoring.svc.cluster.local:9115
|
||||
targets:
|
||||
staticConfig:
|
||||
static:
|
||||
- http://my-service.my-namespace.svc.cluster.local:8080/health
|
||||
```
|
||||
|
||||
|
||||
6. CREA PROBE HTTP/HTTPS PER INGRESS O URL ESTERNI
|
||||
==================================================
|
||||
|
||||
FILE - blackbox-ingress-probes.yaml
|
||||
|
||||
Descrizione:
|
||||
- testa URL esposti tramite Ingress, Gateway, reverse proxy o endpoint pubblici;
|
||||
- usa il modulo generico http_2xx.
|
||||
|
||||
COMANDO - crea blackbox-ingress-probes.yaml
|
||||
```bash
|
||||
cat > blackbox-ingress-probes.yaml <<'EOF'
|
||||
apiVersion: monitoring.coreos.com/v1
|
||||
kind: Probe
|
||||
metadata:
|
||||
name: ingress-http-probe
|
||||
namespace: monitoring
|
||||
labels:
|
||||
release: kube-prometheus-stack
|
||||
spec:
|
||||
jobName: ingress-http-probe
|
||||
interval: 30s
|
||||
scrapeTimeout: 15s
|
||||
module: http_2xx
|
||||
prober:
|
||||
url: blackbox-exporter.monitoring.svc.cluster.local:9115
|
||||
scheme: http
|
||||
path: /probe
|
||||
targets:
|
||||
staticConfig:
|
||||
static:
|
||||
- https://app.example.com/health
|
||||
- https://api.example.com/ready
|
||||
EOF
|
||||
```
|
||||
|
||||
COMANDO - applica Probe endpoint esterni
|
||||
```bash
|
||||
kubectl apply -f blackbox-ingress-probes.yaml
|
||||
```
|
||||
|
||||
|
||||
7. VERIFICA METRICHE IN PROMETHEUS
|
||||
==================================
|
||||
|
||||
COMANDO - apri Prometheus in locale con port-forward
|
||||
```bash
|
||||
kubectl port-forward -n monitoring svc/kube-prometheus-stack-prometheus 9090:9090
|
||||
```
|
||||
|
||||
Poi apri nel browser:
|
||||
|
||||
```text
|
||||
http://localhost:9090
|
||||
```
|
||||
|
||||
QUERY PROMQL - tutte le probe
|
||||
```promql
|
||||
probe_success
|
||||
```
|
||||
|
||||
QUERY PROMQL - probe servizi interni
|
||||
```promql
|
||||
probe_success{job="http-services-probe"}
|
||||
```
|
||||
|
||||
QUERY PROMQL - status code HTTP
|
||||
```promql
|
||||
probe_http_status_code{job="http-services-probe"}
|
||||
```
|
||||
|
||||
QUERY PROMQL - durata probe
|
||||
```promql
|
||||
probe_duration_seconds{job="http-services-probe"}
|
||||
```
|
||||
|
||||
QUERY PROMQL - target falliti
|
||||
```promql
|
||||
probe_success{job="http-services-probe"} == 0
|
||||
```
|
||||
|
||||
Interpretazione:
|
||||
- probe_success = 1: test riuscito.
|
||||
- probe_success = 0: test fallito.
|
||||
- probe_http_status_code: status HTTP ricevuto.
|
||||
- probe_duration_seconds: durata della probe.
|
||||
|
||||
|
||||
8. CREA DASHBOARD GRAFANA PER LE PROBE
|
||||
======================================
|
||||
|
||||
Obiettivo:
|
||||
- visualizzare lo stato delle probe HTTP;
|
||||
- vedere quali target sono UP o DOWN;
|
||||
- controllare status code e latenza per ogni endpoint;
|
||||
- filtrare la dashboard per job e target.
|
||||
|
||||
COMANDO - apri Grafana in locale con port-forward
|
||||
```bash
|
||||
kubectl port-forward -n monitoring svc/kube-prometheus-stack-grafana 3000:80
|
||||
```
|
||||
|
||||
Poi apri nel browser:
|
||||
|
||||
```text
|
||||
http://localhost:3000
|
||||
```
|
||||
|
||||
COMANDO - recupera password admin Grafana se non la conosci
|
||||
```bash
|
||||
kubectl get secret -n monitoring kube-prometheus-stack-grafana \
|
||||
-o jsonpath="{.data.admin-password}" | base64 -d
|
||||
```
|
||||
|
||||
Credenziali predefinite tipiche:
|
||||
```text
|
||||
utente: admin
|
||||
password: valore recuperato dal secret
|
||||
```
|
||||
|
||||
NOTA - datasource Prometheus
|
||||
Con kube-prometheus-stack il datasource Prometheus di solito e' gia' configurato in Grafana.
|
||||
Verifica da Grafana:
|
||||
|
||||
```text
|
||||
Connections -> Data sources -> Prometheus
|
||||
```
|
||||
|
||||
CREAZIONE DASHBOARD DA INTERFACCIA
|
||||
|
||||
1. Vai su Dashboards -> New -> New dashboard.
|
||||
2. Clicca Add visualization.
|
||||
3. Seleziona il datasource Prometheus.
|
||||
4. Crea i pannelli usando le query sotto.
|
||||
5. Salva la dashboard con nome, ad esempio:
|
||||
|
||||
```text
|
||||
Blackbox HTTP Probes
|
||||
```
|
||||
|
||||
VARIABILI CONSIGLIATE
|
||||
|
||||
Variabile job:
|
||||
```text
|
||||
Name: job
|
||||
Type: Query
|
||||
Data source: Prometheus
|
||||
Query: label_values(probe_success, job)
|
||||
Multi-value: enabled
|
||||
Include All option: enabled
|
||||
```
|
||||
|
||||
Variabile instance:
|
||||
```text
|
||||
Name: instance
|
||||
Type: Query
|
||||
Data source: Prometheus
|
||||
Query: label_values(probe_success{job=~"$job"}, instance)
|
||||
Multi-value: enabled
|
||||
Include All option: enabled
|
||||
```
|
||||
|
||||
PANNELLO - stato generale probe
|
||||
|
||||
Tipo pannello: Stat
|
||||
|
||||
Query:
|
||||
```promql
|
||||
min(probe_success{job=~"$job", instance=~"$instance"})
|
||||
```
|
||||
|
||||
Configurazione consigliata:
|
||||
```text
|
||||
Unit: none
|
||||
Thresholds:
|
||||
0 = red
|
||||
1 = green
|
||||
Value mappings:
|
||||
0 -> DOWN
|
||||
1 -> UP
|
||||
```
|
||||
|
||||
PANNELLO - stato per target
|
||||
|
||||
Tipo pannello: State timeline oppure Table
|
||||
|
||||
Query:
|
||||
```promql
|
||||
probe_success{job=~"$job", instance=~"$instance"}
|
||||
```
|
||||
|
||||
Configurazione consigliata:
|
||||
```text
|
||||
Legend: {{ instance }}
|
||||
Value mappings:
|
||||
0 -> DOWN
|
||||
1 -> UP
|
||||
```
|
||||
|
||||
PANNELLO - target attualmente falliti
|
||||
|
||||
Tipo pannello: Table
|
||||
|
||||
Query:
|
||||
```promql
|
||||
probe_success{job=~"$job", instance=~"$instance"} == 0
|
||||
```
|
||||
|
||||
Configurazione consigliata:
|
||||
```text
|
||||
Legend: {{ instance }}
|
||||
Mostra colonne: instance, job, value
|
||||
```
|
||||
|
||||
PANNELLO - durata probe per target
|
||||
|
||||
Tipo pannello: Time series
|
||||
|
||||
Query:
|
||||
```promql
|
||||
probe_duration_seconds{job=~"$job", instance=~"$instance"}
|
||||
```
|
||||
|
||||
Configurazione consigliata:
|
||||
```text
|
||||
Unit: seconds
|
||||
Legend: {{ instance }}
|
||||
Thresholds:
|
||||
1 = yellow
|
||||
2 = red
|
||||
```
|
||||
|
||||
PANNELLO - status code HTTP
|
||||
|
||||
Tipo pannello: Time series oppure Table
|
||||
|
||||
Query:
|
||||
```promql
|
||||
probe_http_status_code{job=~"$job", instance=~"$instance"}
|
||||
```
|
||||
|
||||
Configurazione consigliata:
|
||||
```text
|
||||
Unit: none
|
||||
Legend: {{ instance }}
|
||||
```
|
||||
|
||||
PANNELLO - percentuale disponibilita' per target
|
||||
|
||||
Tipo pannello: Bar gauge oppure Table
|
||||
|
||||
Query:
|
||||
```promql
|
||||
avg_over_time(probe_success{job=~"$job", instance=~"$instance"}[24h]) * 100
|
||||
```
|
||||
|
||||
Configurazione consigliata:
|
||||
```text
|
||||
Unit: percent
|
||||
Min: 0
|
||||
Max: 100
|
||||
Legend: {{ instance }}
|
||||
Thresholds:
|
||||
95 = yellow
|
||||
99 = green
|
||||
```
|
||||
|
||||
PANNELLO - durata media nelle ultime 24 ore
|
||||
|
||||
Tipo pannello: Bar gauge oppure Table
|
||||
|
||||
Query:
|
||||
```promql
|
||||
avg_over_time(probe_duration_seconds{job=~"$job", instance=~"$instance"}[24h])
|
||||
```
|
||||
|
||||
Configurazione consigliata:
|
||||
```text
|
||||
Unit: seconds
|
||||
Legend: {{ instance }}
|
||||
```
|
||||
|
||||
QUERY RAPIDE SENZA VARIABILI
|
||||
|
||||
Se vuoi creare una dashboard solo per la Probe di esempio dei servizi interni:
|
||||
|
||||
```promql
|
||||
probe_success{job="http-services-probe"}
|
||||
probe_http_status_code{job="http-services-probe"}
|
||||
probe_duration_seconds{job="http-services-probe"}
|
||||
probe_success{job="http-services-probe"} == 0
|
||||
avg_over_time(probe_success{job="http-services-probe"}[24h]) * 100
|
||||
```
|
||||
|
||||
Per la Probe di esempio degli Ingress o URL esterni:
|
||||
|
||||
```promql
|
||||
probe_success{job="ingress-http-probe"}
|
||||
probe_http_status_code{job="ingress-http-probe"}
|
||||
probe_duration_seconds{job="ingress-http-probe"}
|
||||
probe_success{job="ingress-http-probe"} == 0
|
||||
avg_over_time(probe_success{job="ingress-http-probe"}[24h]) * 100
|
||||
```
|
||||
|
||||
|
||||
9. CREA ALERT PROMETHEUSRULE
|
||||
============================
|
||||
|
||||
FILE - blackbox-http-alerts.yaml
|
||||
|
||||
Descrizione:
|
||||
- genera alert quando un target HTTP non risponde correttamente;
|
||||
- genera alert quando un target risponde troppo lentamente.
|
||||
|
||||
COMANDO - crea blackbox-http-alerts.yaml
|
||||
```bash
|
||||
cat > blackbox-http-alerts.yaml <<'EOF'
|
||||
apiVersion: monitoring.coreos.com/v1
|
||||
kind: PrometheusRule
|
||||
metadata:
|
||||
name: blackbox-http-alerts
|
||||
namespace: monitoring
|
||||
labels:
|
||||
release: kube-prometheus-stack
|
||||
spec:
|
||||
groups:
|
||||
- name: blackbox-http
|
||||
rules:
|
||||
- alert: BlackboxHttpProbeFailed
|
||||
expr: probe_success == 0
|
||||
for: 2m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "HTTP probe fallita"
|
||||
description: "Il target {{ $labels.instance }} non risponde correttamente da almeno 2 minuti."
|
||||
|
||||
- alert: BlackboxHttpSlowResponse
|
||||
expr: probe_duration_seconds > 2
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "HTTP probe lenta"
|
||||
description: "Il target {{ $labels.instance }} risponde in piu' di 2 secondi da almeno 5 minuti."
|
||||
EOF
|
||||
```
|
||||
|
||||
COMANDO - applica alert
|
||||
```bash
|
||||
kubectl apply -f blackbox-http-alerts.yaml
|
||||
```
|
||||
|
||||
COMANDO - verifica PrometheusRule
|
||||
```bash
|
||||
kubectl get prometheusrule -n monitoring | grep blackbox
|
||||
kubectl describe prometheusrule blackbox-http-alerts -n monitoring
|
||||
```
|
||||
|
||||
|
||||
10. TEST MANUALE BLACKBOX EXPORTER
|
||||
=================================
|
||||
|
||||
COMANDO - port-forward blackbox-exporter
|
||||
```bash
|
||||
kubectl port-forward -n monitoring svc/blackbox-exporter 9115:9115
|
||||
```
|
||||
|
||||
COMANDO - esegui probe manuale da un altro terminale
|
||||
```bash
|
||||
curl "http://localhost:9115/probe?target=http://my-service.my-namespace.svc.cluster.local:8080/health&module=http_k8s_health"
|
||||
```
|
||||
|
||||
OUTPUT ATTESO - metriche indicative
|
||||
```text
|
||||
probe_success 1
|
||||
probe_http_status_code 200
|
||||
```
|
||||
|
||||
|
||||
11. TROUBLESHOOTING
|
||||
===================
|
||||
|
||||
CASO - Prometheus non vede la Probe
|
||||
|
||||
COMANDO
|
||||
```bash
|
||||
kubectl get probe -A
|
||||
kubectl describe probe -n monitoring http-services-probe
|
||||
kubectl get prometheus -n monitoring -o yaml
|
||||
```
|
||||
|
||||
Controlla:
|
||||
- spec.probeSelector
|
||||
- spec.probeNamespaceSelector
|
||||
- metadata.labels della Probe
|
||||
|
||||
CASO - blackbox-exporter non parte
|
||||
|
||||
COMANDO
|
||||
```bash
|
||||
kubectl get pods -n monitoring | grep blackbox
|
||||
kubectl logs -n monitoring deploy/blackbox-exporter
|
||||
```
|
||||
|
||||
CASO - servizio target non raggiungibile
|
||||
|
||||
COMANDO
|
||||
```bash
|
||||
kubectl get svc -n my-namespace
|
||||
kubectl get endpoints -n my-namespace my-service
|
||||
kubectl describe svc -n my-namespace my-service
|
||||
```
|
||||
|
||||
CASO - DNS Kubernetes non risolve
|
||||
|
||||
COMANDO
|
||||
```bash
|
||||
kubectl run dns-test --rm -it --image=busybox:1.36 --restart=Never -- nslookup my-service.my-namespace.svc.cluster.local
|
||||
```
|
||||
|
||||
CASO - endpoint HTTPS fallisce per certificati, redirect o header
|
||||
|
||||
Azioni consigliate:
|
||||
- usare il modulo http_2xx per endpoint esterni generici;
|
||||
- verificare se l'endpoint richiede SNI, autenticazione, header custom o path diverso;
|
||||
- aggiungere un modulo dedicato in blackbox-values.yaml se serve una configurazione HTTP specifica.
|
||||
|
||||
|
||||
12. ORDINE DI ESECUZIONE CONSIGLIATO
|
||||
====================================
|
||||
|
||||
COMANDO - sequenza completa
|
||||
```bash
|
||||
helm repo add prometheus-community https://prometheus-community.github.io/helm-charts
|
||||
helm repo update
|
||||
|
||||
helm upgrade --install blackbox-exporter prometheus-community/prometheus-blackbox-exporter \
|
||||
--namespace monitoring \
|
||||
--create-namespace \
|
||||
-f blackbox-values.yaml
|
||||
|
||||
kubectl apply -f blackbox-probes.yaml
|
||||
kubectl apply -f blackbox-ingress-probes.yaml
|
||||
kubectl apply -f blackbox-http-alerts.yaml
|
||||
```
|
||||
|
||||
FILE CONSIGLIATI DA VERSIONARE
|
||||
```text
|
||||
blackbox-values.yaml
|
||||
blackbox-probes.yaml
|
||||
blackbox-ingress-probes.yaml
|
||||
blackbox-http-alerts.yaml
|
||||
```
|
||||
|
||||
|
||||
13. NOTE PER DEV, QA E PROD
|
||||
===========================
|
||||
|
||||
Per separare gli ambienti e semplificare dashboard e alert, usa Probe distinte:
|
||||
|
||||
```text
|
||||
dev-http-services-probe
|
||||
qa-http-services-probe
|
||||
prod-http-services-probe
|
||||
```
|
||||
|
||||
QUERY PROMQL - ambiente dev
|
||||
```promql
|
||||
probe_success{job="dev-http-services-probe"}
|
||||
```
|
||||
|
||||
QUERY PROMQL - target prod falliti
|
||||
```promql
|
||||
probe_success{job="prod-http-services-probe"} == 0
|
||||
```
|
||||
|
||||
Per endpoint critici in produzione, valuta:
|
||||
- interval piu' basso, ad esempio 15s o 30s;
|
||||
- durata for degli alert tra 1m e 5m in base alla criticita';
|
||||
- dashboard Grafana con stato, status code e latenza per target.
|
||||
@@ -0,0 +1,553 @@
|
||||
apiVersion: monitoring.coreos.com/v1
|
||||
kind: Probe
|
||||
metadata:
|
||||
name: <namespace>-<env>-<endpoint>-probe
|
||||
namespace: monitoring
|
||||
labels:
|
||||
release: kube-prometheus-stack
|
||||
spec:
|
||||
jobName: <namespace>-<env>-<endpoint>-probe
|
||||
interval: 30s
|
||||
scrapeTimeout: 15s
|
||||
module: http_2xx
|
||||
prober:
|
||||
url: blackbox-exporter.monitoring.svc.cluster.local:9115
|
||||
scheme: http
|
||||
path: /probe
|
||||
targets:
|
||||
staticConfig:
|
||||
static:
|
||||
- https://<endpoint>
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
name: <namespace>-<env>-<endpoint>-dashboard
|
||||
namespace: monitoring
|
||||
labels:
|
||||
grafana_dashboard: "1"
|
||||
data:
|
||||
<namespace>-<env>-<endpoint>-dashboard.json: |
|
||||
{
|
||||
"annotations": {
|
||||
"list": []
|
||||
},
|
||||
"editable": true,
|
||||
"fiscalYearStartMonth": 0,
|
||||
"graphTooltip": 1,
|
||||
"links": [],
|
||||
"panels": [
|
||||
{
|
||||
"type": "stat",
|
||||
"title": "Services UP",
|
||||
"id": 1,
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 6,
|
||||
"x": 0,
|
||||
"y": 0
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "sum(probe_success{job=\"<namespace>-<env>-<endpoint>-probe\",instance=~\"$instance\"})"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "short",
|
||||
"thresholds": {
|
||||
"mode": "absolute",
|
||||
"steps": [
|
||||
{
|
||||
"color": "red",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"color": "green",
|
||||
"value": 1
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
"options": {
|
||||
"reduceOptions": {
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
],
|
||||
"fields": "",
|
||||
"values": false
|
||||
},
|
||||
"orientation": "auto",
|
||||
"textMode": "auto",
|
||||
"colorMode": "value",
|
||||
"graphMode": "none"
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "stat",
|
||||
"title": "Services DOWN",
|
||||
"id": 2,
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 6,
|
||||
"x": 6,
|
||||
"y": 0
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "count(probe_success{job=\"<namespace>-<env>-<endpoint>-probe\",instance=~\"$instance\"}) - sum(probe_success{job=\"<namespace>-<env>-<endpoint>-probe\",instance=~\"$instance\"})"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "short",
|
||||
"thresholds": {
|
||||
"mode": "absolute",
|
||||
"steps": [
|
||||
{
|
||||
"color": "green",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"color": "red",
|
||||
"value": 1
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
"options": {
|
||||
"reduceOptions": {
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
],
|
||||
"fields": "",
|
||||
"values": false
|
||||
},
|
||||
"orientation": "auto",
|
||||
"textMode": "auto",
|
||||
"colorMode": "value",
|
||||
"graphMode": "none"
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "stat",
|
||||
"title": "Availability 24h",
|
||||
"id": 3,
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 6,
|
||||
"x": 12,
|
||||
"y": 0
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "avg(avg_over_time(probe_success{job=\"<namespace>-<env>-<endpoint>-probe\",instance=~\"$instance\"}[24h])) * 100"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "percent",
|
||||
"min": 0,
|
||||
"max": 100,
|
||||
"thresholds": {
|
||||
"mode": "absolute",
|
||||
"steps": [
|
||||
{
|
||||
"color": "red",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"color": "orange",
|
||||
"value": 99
|
||||
},
|
||||
{
|
||||
"color": "green",
|
||||
"value": 99.9
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
"options": {
|
||||
"reduceOptions": {
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
],
|
||||
"fields": "",
|
||||
"values": false
|
||||
},
|
||||
"orientation": "auto",
|
||||
"textMode": "auto",
|
||||
"colorMode": "value",
|
||||
"graphMode": "area"
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "stat",
|
||||
"title": "Avg Latency",
|
||||
"id": 4,
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 6,
|
||||
"x": 18,
|
||||
"y": 0
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "avg(probe_duration_seconds{job=\"<namespace>-<env>-<endpoint>-probe\",instance=~\"$instance\"}) * 1000"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "ms",
|
||||
"decimals": 0
|
||||
}
|
||||
},
|
||||
"options": {
|
||||
"reduceOptions": {
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
],
|
||||
"fields": "",
|
||||
"values": false
|
||||
},
|
||||
"orientation": "auto",
|
||||
"textMode": "auto",
|
||||
"colorMode": "value",
|
||||
"graphMode": "none"
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "table",
|
||||
"title": "HTTP Services Status",
|
||||
"id": 5,
|
||||
"gridPos": {
|
||||
"h": 8,
|
||||
"w": 12,
|
||||
"x": 0,
|
||||
"y": 5
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "probe_success{job=\"<namespace>-<env>-<endpoint>-probe\",instance=~\"$instance\"}",
|
||||
"instant": true,
|
||||
"format": "table"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"custom": {
|
||||
"align": "auto",
|
||||
"cellOptions": {
|
||||
"type": "color-text"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"transformations": [
|
||||
{
|
||||
"id": "organize",
|
||||
"options": {
|
||||
"excludeByName": {
|
||||
"Time": true
|
||||
},
|
||||
"renameByName": {
|
||||
"Value": "Status",
|
||||
"instance": "Service"
|
||||
}
|
||||
}
|
||||
}
|
||||
],
|
||||
"options": {
|
||||
"showHeader": true,
|
||||
"cellHeight": "sm"
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "table",
|
||||
"title": "HTTP Status Code",
|
||||
"id": 6,
|
||||
"gridPos": {
|
||||
"h": 8,
|
||||
"w": 12,
|
||||
"x": 12,
|
||||
"y": 5
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "probe_http_status_code{job=\"<namespace>-<env>-<endpoint>-probe\",instance=~\"$instance\"}",
|
||||
"instant": true,
|
||||
"format": "table"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "short",
|
||||
"custom": {
|
||||
"align": "auto",
|
||||
"cellOptions": {
|
||||
"type": "color-text"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"transformations": [
|
||||
{
|
||||
"id": "organize",
|
||||
"options": {
|
||||
"excludeByName": {
|
||||
"Time": true
|
||||
},
|
||||
"renameByName": {
|
||||
"Value": "HTTP Status",
|
||||
"instance": "Service"
|
||||
}
|
||||
}
|
||||
}
|
||||
],
|
||||
"options": {
|
||||
"showHeader": true,
|
||||
"cellHeight": "sm"
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "timeseries",
|
||||
"title": "HTTP Response Time",
|
||||
"id": 7,
|
||||
"gridPos": {
|
||||
"h": 9,
|
||||
"w": 24,
|
||||
"x": 0,
|
||||
"y": 13
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "probe_duration_seconds{job=\"<namespace>-<env>-<endpoint>-probe\",instance=~\"$instance\"} * 1000",
|
||||
"legendFormat": "{{instance}}"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "ms",
|
||||
"decimals": 0
|
||||
}
|
||||
},
|
||||
"options": {
|
||||
"legend": {
|
||||
"displayMode": "list",
|
||||
"placement": "bottom"
|
||||
},
|
||||
"tooltip": {
|
||||
"mode": "multi",
|
||||
"sort": "desc"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "table",
|
||||
"title": "Currently DOWN",
|
||||
"id": 8,
|
||||
"gridPos": {
|
||||
"h": 7,
|
||||
"w": 12,
|
||||
"x": 0,
|
||||
"y": 22
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "probe_success{job=\"<namespace>-<env>-<endpoint>-probe\",instance=~\"$instance\"} == 0",
|
||||
"instant": true,
|
||||
"format": "table"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"custom": {
|
||||
"cellOptions": {
|
||||
"type": "color-text"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"transformations": [
|
||||
{
|
||||
"id": "organize",
|
||||
"options": {
|
||||
"excludeByName": {
|
||||
"Time": true
|
||||
},
|
||||
"renameByName": {
|
||||
"Value": "Status",
|
||||
"instance": "Service"
|
||||
}
|
||||
}
|
||||
}
|
||||
],
|
||||
"options": {
|
||||
"showHeader": true,
|
||||
"cellHeight": "sm"
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "timeseries",
|
||||
"title": "Availability 24h",
|
||||
"id": 9,
|
||||
"gridPos": {
|
||||
"h": 9,
|
||||
"w": 12,
|
||||
"x": 12,
|
||||
"y": 22
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "avg_over_time(probe_success{job=\"<namespace>-<env>-<endpoint>-probe\",instance=~\"$instance\"}[24h]) * 100",
|
||||
"legendFormat": "{{instance}}"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "percent",
|
||||
"min": 0,
|
||||
"max": 100,
|
||||
"decimals": 2
|
||||
}
|
||||
},
|
||||
"options": {
|
||||
"legend": {
|
||||
"displayMode": "list",
|
||||
"placement": "bottom"
|
||||
},
|
||||
"tooltip": {
|
||||
"mode": "multi",
|
||||
"sort": "desc"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "timeseries",
|
||||
"title": "P95 Latency - 1h",
|
||||
"id": 10,
|
||||
"gridPos": {
|
||||
"h": 9,
|
||||
"w": 24,
|
||||
"x": 0,
|
||||
"y": 31
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "quantile_over_time(0.95, probe_duration_seconds{job=\"<namespace>-<env>-<endpoint>-probe\",instance=~\"$instance\"}[1h]) * 1000",
|
||||
"legendFormat": "{{instance}}"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "ms",
|
||||
"decimals": 0
|
||||
}
|
||||
},
|
||||
"options": {
|
||||
"legend": {
|
||||
"displayMode": "list",
|
||||
"placement": "bottom"
|
||||
},
|
||||
"tooltip": {
|
||||
"mode": "multi",
|
||||
"sort": "desc"
|
||||
}
|
||||
}
|
||||
}
|
||||
],
|
||||
"refresh": "30s",
|
||||
"schemaVersion": 41,
|
||||
"tags": [
|
||||
"blackbox",
|
||||
"http",
|
||||
"monitoring"
|
||||
],
|
||||
|
||||
"templating": {
|
||||
"list": [
|
||||
{
|
||||
"name": "instance",
|
||||
"label": "Service",
|
||||
"type": "query",
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"definition": "label_values(probe_success{job=\"<namespace>-<env>-<endpoint>-probe\"}, instance)",
|
||||
"query": {
|
||||
"query": "label_values(probe_success{job=\"<namespace>-<env>-<endpoint>-probe\"}, instance)",
|
||||
"refId": "StandardVariableQuery"
|
||||
},
|
||||
"includeAll": true,
|
||||
"multi": true,
|
||||
"allValue": ".*",
|
||||
"refresh": 1,
|
||||
"sort": 1
|
||||
}
|
||||
]
|
||||
},
|
||||
"time": {
|
||||
"from": "now-24h",
|
||||
"to": "now"
|
||||
},
|
||||
"timepicker": {},
|
||||
"timezone": "browser",
|
||||
"title": "<namespace>-<env>-<endpoint>-dashboard",
|
||||
"uid": "<namespace>-<env>-<endpoint>-dashboard",
|
||||
"version": 1,
|
||||
"weekStart": ""
|
||||
}
|
||||
@@ -0,0 +1,553 @@
|
||||
apiVersion: monitoring.coreos.com/v1
|
||||
kind: Probe
|
||||
metadata:
|
||||
name: <my-http-services-probe>
|
||||
namespace: monitoring
|
||||
labels:
|
||||
release: kube-prometheus-stack
|
||||
spec:
|
||||
jobName: <my-http-services-probe>
|
||||
interval: 30s
|
||||
scrapeTimeout: 15s
|
||||
module: http_k8s_health
|
||||
prober:
|
||||
url: blackbox-exporter.monitoring.svc.cluster.local:9115
|
||||
scheme: http
|
||||
path: /probe
|
||||
targets:
|
||||
staticConfig:
|
||||
static:
|
||||
- http://<my-service>.<my-namespace>.svc.cluster.local:<my-port>/health
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
name: <my-http-services-dashboard>
|
||||
namespace: monitoring
|
||||
labels:
|
||||
grafana_dashboard: "1"
|
||||
data:
|
||||
<my-http-services-dashboard>.json: |
|
||||
{
|
||||
"annotations": {
|
||||
"list": []
|
||||
},
|
||||
"editable": true,
|
||||
"fiscalYearStartMonth": 0,
|
||||
"graphTooltip": 1,
|
||||
"links": [],
|
||||
"panels": [
|
||||
{
|
||||
"type": "stat",
|
||||
"title": "Services UP",
|
||||
"id": 1,
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 6,
|
||||
"x": 0,
|
||||
"y": 0
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "sum(probe_success{job=\"<my-http-services-probe>\",instance=~\"$instance\"})"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "short",
|
||||
"thresholds": {
|
||||
"mode": "absolute",
|
||||
"steps": [
|
||||
{
|
||||
"color": "red",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"color": "green",
|
||||
"value": 1
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
"options": {
|
||||
"reduceOptions": {
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
],
|
||||
"fields": "",
|
||||
"values": false
|
||||
},
|
||||
"orientation": "auto",
|
||||
"textMode": "auto",
|
||||
"colorMode": "value",
|
||||
"graphMode": "none"
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "stat",
|
||||
"title": "Services DOWN",
|
||||
"id": 2,
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 6,
|
||||
"x": 6,
|
||||
"y": 0
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "count(probe_success{job=\"<my-http-services-probe>\",instance=~\"$instance\"}) - sum(probe_success{job=\"<my-http-services-probe>\",instance=~\"$instance\"})"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "short",
|
||||
"thresholds": {
|
||||
"mode": "absolute",
|
||||
"steps": [
|
||||
{
|
||||
"color": "green",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"color": "red",
|
||||
"value": 1
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
"options": {
|
||||
"reduceOptions": {
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
],
|
||||
"fields": "",
|
||||
"values": false
|
||||
},
|
||||
"orientation": "auto",
|
||||
"textMode": "auto",
|
||||
"colorMode": "value",
|
||||
"graphMode": "none"
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "stat",
|
||||
"title": "Availability 24h",
|
||||
"id": 3,
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 6,
|
||||
"x": 12,
|
||||
"y": 0
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "avg(avg_over_time(probe_success{job=\"<my-http-services-probe>\",instance=~\"$instance\"}[24h])) * 100"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "percent",
|
||||
"min": 0,
|
||||
"max": 100,
|
||||
"thresholds": {
|
||||
"mode": "absolute",
|
||||
"steps": [
|
||||
{
|
||||
"color": "red",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"color": "orange",
|
||||
"value": 99
|
||||
},
|
||||
{
|
||||
"color": "green",
|
||||
"value": 99.9
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
"options": {
|
||||
"reduceOptions": {
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
],
|
||||
"fields": "",
|
||||
"values": false
|
||||
},
|
||||
"orientation": "auto",
|
||||
"textMode": "auto",
|
||||
"colorMode": "value",
|
||||
"graphMode": "area"
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "stat",
|
||||
"title": "Avg Latency",
|
||||
"id": 4,
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 6,
|
||||
"x": 18,
|
||||
"y": 0
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "avg(probe_duration_seconds{job=\"<my-http-services-probe>\",instance=~\"$instance\"}) * 1000"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "ms",
|
||||
"decimals": 0
|
||||
}
|
||||
},
|
||||
"options": {
|
||||
"reduceOptions": {
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
],
|
||||
"fields": "",
|
||||
"values": false
|
||||
},
|
||||
"orientation": "auto",
|
||||
"textMode": "auto",
|
||||
"colorMode": "value",
|
||||
"graphMode": "none"
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "table",
|
||||
"title": "HTTP Services Status",
|
||||
"id": 5,
|
||||
"gridPos": {
|
||||
"h": 8,
|
||||
"w": 12,
|
||||
"x": 0,
|
||||
"y": 5
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "probe_success{job=\"<my-http-services-probe>\",instance=~\"$instance\"}",
|
||||
"instant": true,
|
||||
"format": "table"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"custom": {
|
||||
"align": "auto",
|
||||
"cellOptions": {
|
||||
"type": "color-text"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"transformations": [
|
||||
{
|
||||
"id": "organize",
|
||||
"options": {
|
||||
"excludeByName": {
|
||||
"Time": true
|
||||
},
|
||||
"renameByName": {
|
||||
"Value": "Status",
|
||||
"instance": "Service"
|
||||
}
|
||||
}
|
||||
}
|
||||
],
|
||||
"options": {
|
||||
"showHeader": true,
|
||||
"cellHeight": "sm"
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "table",
|
||||
"title": "HTTP Status Code",
|
||||
"id": 6,
|
||||
"gridPos": {
|
||||
"h": 8,
|
||||
"w": 12,
|
||||
"x": 12,
|
||||
"y": 5
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "probe_http_status_code{job=\"<my-http-services-probe>\",instance=~\"$instance\"}",
|
||||
"instant": true,
|
||||
"format": "table"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "short",
|
||||
"custom": {
|
||||
"align": "auto",
|
||||
"cellOptions": {
|
||||
"type": "color-text"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"transformations": [
|
||||
{
|
||||
"id": "organize",
|
||||
"options": {
|
||||
"excludeByName": {
|
||||
"Time": true
|
||||
},
|
||||
"renameByName": {
|
||||
"Value": "HTTP Status",
|
||||
"instance": "Service"
|
||||
}
|
||||
}
|
||||
}
|
||||
],
|
||||
"options": {
|
||||
"showHeader": true,
|
||||
"cellHeight": "sm"
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "timeseries",
|
||||
"title": "HTTP Response Time",
|
||||
"id": 7,
|
||||
"gridPos": {
|
||||
"h": 9,
|
||||
"w": 24,
|
||||
"x": 0,
|
||||
"y": 13
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "probe_duration_seconds{job=\"<my-http-services-probe>\",instance=~\"$instance\"} * 1000",
|
||||
"legendFormat": "{{instance}}"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "ms",
|
||||
"decimals": 0
|
||||
}
|
||||
},
|
||||
"options": {
|
||||
"legend": {
|
||||
"displayMode": "list",
|
||||
"placement": "bottom"
|
||||
},
|
||||
"tooltip": {
|
||||
"mode": "multi",
|
||||
"sort": "desc"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "table",
|
||||
"title": "Currently DOWN",
|
||||
"id": 8,
|
||||
"gridPos": {
|
||||
"h": 7,
|
||||
"w": 12,
|
||||
"x": 0,
|
||||
"y": 22
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "probe_success{job=\"<my-http-services-probe>\",instance=~\"$instance\"} == 0",
|
||||
"instant": true,
|
||||
"format": "table"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"custom": {
|
||||
"cellOptions": {
|
||||
"type": "color-text"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"transformations": [
|
||||
{
|
||||
"id": "organize",
|
||||
"options": {
|
||||
"excludeByName": {
|
||||
"Time": true
|
||||
},
|
||||
"renameByName": {
|
||||
"Value": "Status",
|
||||
"instance": "Service"
|
||||
}
|
||||
}
|
||||
}
|
||||
],
|
||||
"options": {
|
||||
"showHeader": true,
|
||||
"cellHeight": "sm"
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "timeseries",
|
||||
"title": "Availability 24h",
|
||||
"id": 9,
|
||||
"gridPos": {
|
||||
"h": 9,
|
||||
"w": 12,
|
||||
"x": 12,
|
||||
"y": 22
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "avg_over_time(probe_success{job=\"<my-http-services-probe>\",instance=~\"$instance\"}[24h]) * 100",
|
||||
"legendFormat": "{{instance}}"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "percent",
|
||||
"min": 0,
|
||||
"max": 100,
|
||||
"decimals": 2
|
||||
}
|
||||
},
|
||||
"options": {
|
||||
"legend": {
|
||||
"displayMode": "list",
|
||||
"placement": "bottom"
|
||||
},
|
||||
"tooltip": {
|
||||
"mode": "multi",
|
||||
"sort": "desc"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "timeseries",
|
||||
"title": "P95 Latency - 1h",
|
||||
"id": 10,
|
||||
"gridPos": {
|
||||
"h": 9,
|
||||
"w": 24,
|
||||
"x": 0,
|
||||
"y": 31
|
||||
},
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"refId": "A",
|
||||
"expr": "quantile_over_time(0.95, probe_duration_seconds{job=\"<my-http-services-probe>\",instance=~\"$instance\"}[1h]) * 1000",
|
||||
"legendFormat": "{{instance}}"
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"unit": "ms",
|
||||
"decimals": 0
|
||||
}
|
||||
},
|
||||
"options": {
|
||||
"legend": {
|
||||
"displayMode": "list",
|
||||
"placement": "bottom"
|
||||
},
|
||||
"tooltip": {
|
||||
"mode": "multi",
|
||||
"sort": "desc"
|
||||
}
|
||||
}
|
||||
}
|
||||
],
|
||||
"refresh": "30s",
|
||||
"schemaVersion": 41,
|
||||
"tags": [
|
||||
"blackbox",
|
||||
"http",
|
||||
"monitoring"
|
||||
],
|
||||
|
||||
"templating": {
|
||||
"list": [
|
||||
{
|
||||
"name": "instance",
|
||||
"label": "Service",
|
||||
"type": "query",
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "prometheus"
|
||||
},
|
||||
"definition": "label_values(probe_success{job=\"<my-http-services-probe>\"}, instance)",
|
||||
"query": {
|
||||
"query": "label_values(probe_success{job=\"<my-http-services-probe>\"}, instance)",
|
||||
"refId": "StandardVariableQuery"
|
||||
},
|
||||
"includeAll": true,
|
||||
"multi": true,
|
||||
"allValue": ".*",
|
||||
"refresh": 1,
|
||||
"sort": 1
|
||||
}
|
||||
]
|
||||
},
|
||||
"time": {
|
||||
"from": "now-24h",
|
||||
"to": "now"
|
||||
},
|
||||
"timepicker": {},
|
||||
"timezone": "browser",
|
||||
"title": "<my-http-services-dashboard>",
|
||||
"uid": "<my-http-services-dashboard>",
|
||||
"version": 1,
|
||||
"weekStart": ""
|
||||
}
|
||||
Reference in New Issue
Block a user