{
  "__inputs": [
    { "name": "DS_PROMETHEUS", "label": "Prometheus / Mimir", "description": "Where your Proxmox metrics land", "type": "datasource", "pluginId": "prometheus", "pluginName": "Prometheus" },
    { "name": "DS_LOKI", "label": "Loki", "description": "Where your Proxmox journal lands", "type": "datasource", "pluginId": "loki", "pluginName": "Loki" }
  ],
  "__requires": [
    { "type": "grafana", "id": "grafana", "name": "Grafana", "version": "11.0.0" },
    { "type": "datasource", "id": "prometheus", "name": "Prometheus", "version": "1.0.0" },
    { "type": "datasource", "id": "loki", "name": "Loki", "version": "1.0.0" }
  ],
  "annotations": { "list": [] },
  "editable": true,
  "graphTooltip": 1,
  "time": { "from": "now-6h", "to": "now" },
  "timezone": "",
  "title": "Proxmox VE — OpenTelemetry",
  "tags": ["proxmox", "opentelemetry", "linkmesh"],
  "schemaVersion": 39,
  "version": 1,
  "templating": {
    "list": [
      {
        "name": "host",
        "label": "Node",
        "type": "query",
        "datasource": { "type": "prometheus", "uid": "${DS_PROMETHEUS}" },
        "definition": "label_values(system_cpu_time_seconds_total, host_name)",
        "query": { "query": "label_values(system_cpu_time_seconds_total, host_name)", "refId": "hostVar" },
        "refresh": 2,
        "sort": 1,
        "includeAll": false,
        "multi": false
      }
    ]
  },
  "panels": [
    {
      "id": 1,
      "title": "CPU busy %",
      "type": "timeseries",
      "datasource": { "type": "prometheus", "uid": "${DS_PROMETHEUS}" },
      "gridPos": { "h": 8, "w": 12, "x": 0, "y": 0 },
      "fieldConfig": { "defaults": { "unit": "percentunit", "min": 0, "max": 1 }, "overrides": [] },
      "targets": [
        { "refId": "A", "expr": "sum(rate(system_cpu_time_seconds_total{host_name=\"$host\", state!=\"idle\"}[5m])) / sum(rate(system_cpu_time_seconds_total{host_name=\"$host\"}[5m]))", "legendFormat": "CPU busy" }
      ]
    },
    {
      "id": 2,
      "title": "IO delay %",
      "description": "CPU time spent waiting on IO — the first number a Proxmox admin checks. Sustained values above ~10% usually mean saturated storage.",
      "type": "timeseries",
      "datasource": { "type": "prometheus", "uid": "${DS_PROMETHEUS}" },
      "gridPos": { "h": 8, "w": 12, "x": 12, "y": 0 },
      "fieldConfig": { "defaults": { "unit": "percentunit", "min": 0, "max": 1 }, "overrides": [] },
      "targets": [
        { "refId": "A", "expr": "sum(rate(system_cpu_time_seconds_total{host_name=\"$host\", state=\"wait\"}[5m])) / sum(rate(system_cpu_time_seconds_total{host_name=\"$host\"}[5m]))", "legendFormat": "IO delay" }
      ]
    },
    {
      "id": 3,
      "title": "Memory used %",
      "description": "Includes buffers/cache; on ZFS nodes the ARC shows up here too.",
      "type": "timeseries",
      "datasource": { "type": "prometheus", "uid": "${DS_PROMETHEUS}" },
      "gridPos": { "h": 8, "w": 12, "x": 0, "y": 8 },
      "fieldConfig": { "defaults": { "unit": "percentunit", "min": 0, "max": 1 }, "overrides": [] },
      "targets": [
        { "refId": "A", "expr": "sum(system_memory_usage_bytes{host_name=\"$host\", state=~\"used|buffered|cached\"}) / sum(system_memory_usage_bytes{host_name=\"$host\"})", "legendFormat": "Memory used" }
      ]
    },
    {
      "id": 4,
      "title": "Filesystem used % (by mount)",
      "type": "timeseries",
      "datasource": { "type": "prometheus", "uid": "${DS_PROMETHEUS}" },
      "gridPos": { "h": 8, "w": 12, "x": 12, "y": 8 },
      "fieldConfig": { "defaults": { "unit": "percentunit", "min": 0, "max": 1 }, "overrides": [] },
      "targets": [
        { "refId": "A", "expr": "sum by (mountpoint) (system_filesystem_usage_bytes{host_name=\"$host\", state=\"used\"}) / sum by (mountpoint) (system_filesystem_usage_bytes{host_name=\"$host\"})", "legendFormat": "{{mountpoint}}" }
      ]
    },
    {
      "id": 5,
      "title": "Proxmox storage used % (native push)",
      "description": "Fed by the Proxmox VE 9 native OTLP metric push. Metric names carry proxmox_* prefixes but can evolve between PVE releases — verify the exact names in Explore against your version and adjust this query if needed.",
      "type": "timeseries",
      "datasource": { "type": "prometheus", "uid": "${DS_PROMETHEUS}" },
      "gridPos": { "h": 8, "w": 12, "x": 0, "y": 16 },
      "fieldConfig": { "defaults": { "unit": "percentunit", "min": 0, "max": 1 }, "overrides": [] },
      "targets": [
        { "refId": "A", "expr": "proxmox_storage_used_bytes / proxmox_storage_total_bytes", "legendFormat": "{{storage}}" }
      ]
    },
    {
      "id": 6,
      "title": "Network throughput",
      "type": "timeseries",
      "datasource": { "type": "prometheus", "uid": "${DS_PROMETHEUS}" },
      "gridPos": { "h": 8, "w": 12, "x": 12, "y": 16 },
      "fieldConfig": { "defaults": { "unit": "Bps" }, "overrides": [] },
      "targets": [
        { "refId": "A", "expr": "sum by (direction) (rate(system_network_io_bytes_total{host_name=\"$host\"}[5m]))", "legendFormat": "{{direction}}" }
      ]
    },
    {
      "id": 7,
      "title": "Node reporting",
      "type": "stat",
      "datasource": { "type": "prometheus", "uid": "${DS_PROMETHEUS}" },
      "gridPos": { "h": 6, "w": 6, "x": 0, "y": 24 },
      "fieldConfig": {
        "defaults": {
          "unit": "short",
          "mappings": [
            { "type": "value", "options": { "0": { "text": "No data", "color": "red" } } },
            { "type": "range", "options": { "from": 1, "to": 1000000, "result": { "text": "Reporting", "color": "green" } } }
          ],
          "thresholds": { "mode": "absolute", "steps": [ { "color": "red", "value": null }, { "color": "green", "value": 1 } ] }
        },
        "overrides": []
      },
      "options": { "reduceOptions": { "calcs": ["lastNotNull"] }, "colorMode": "background", "graphMode": "none" },
      "targets": [
        { "refId": "A", "expr": "count(system_cpu_time_seconds_total{host_name=\"$host\"})", "legendFormat": "reporting" }
      ]
    },
    {
      "id": 8,
      "title": "Ceph health",
      "description": "Requires the prometheus/ceph receiver scraping the Ceph manager nodes (see the guide). 0 = OK, 1 = WARN, 2 = ERR.",
      "type": "stat",
      "datasource": { "type": "prometheus", "uid": "${DS_PROMETHEUS}" },
      "gridPos": { "h": 6, "w": 6, "x": 6, "y": 24 },
      "fieldConfig": {
        "defaults": {
          "unit": "short",
          "mappings": [
            { "type": "value", "options": { "0": { "text": "HEALTH_OK", "color": "green" }, "1": { "text": "HEALTH_WARN", "color": "yellow" }, "2": { "text": "HEALTH_ERR", "color": "red" } } }
          ],
          "thresholds": { "mode": "absolute", "steps": [ { "color": "green", "value": null }, { "color": "yellow", "value": 1 }, { "color": "red", "value": 2 } ] }
        },
        "overrides": []
      },
      "options": { "reduceOptions": { "calcs": ["lastNotNull"] }, "colorMode": "background", "graphMode": "none" },
      "targets": [
        { "refId": "A", "expr": "ceph_health_status", "legendFormat": "ceph" }
      ]
    },
    {
      "id": 9,
      "title": "Backup-related journal errors",
      "description": "Counts vzdump/backup lines with error/fail wording in the journald units this setup collects. An early-warning signal, not a job-level guarantee — full task logs stay in /var/log/pve/tasks; use Proxmox's own notification system for guaranteed job alerts.",
      "type": "timeseries",
      "datasource": { "type": "loki", "uid": "${DS_LOKI}" },
      "gridPos": { "h": 6, "w": 12, "x": 12, "y": 24 },
      "fieldConfig": { "defaults": { "unit": "short" }, "overrides": [] },
      "targets": [
        { "refId": "A", "expr": "sum(count_over_time({host_name=\"$host\"} |~ `(?i)(vzdump|backup)` |~ `(?i)(error|fail)` [$__interval]))", "legendFormat": "backup errors" }
      ]
    },
    {
      "id": 10,
      "title": "Journal",
      "type": "logs",
      "datasource": { "type": "loki", "uid": "${DS_LOKI}" },
      "gridPos": { "h": 10, "w": 24, "x": 0, "y": 30 },
      "options": { "showTime": true, "wrapLogMessage": true, "sortOrder": "Descending" },
      "targets": [
        { "refId": "A", "expr": "{host_name=\"$host\"}" }
      ]
    }
  ]
}
