{
  "title": "Pod CPU Throttling (cgroup)",
  "uid": "servloci-cpu-throttle",
  "tags": [
    "kubernetes",
    "cpu",
    "cgroup",
    "throttling",
    "servloci"
  ],
  "description": "CFS throttling, usage vs limit/request, PSI, steal and Go scheduler latency per pod. https://servloci.in/blog/cpu-throttling/",
  "timezone": "browser",
  "schemaVersion": 39,
  "version": 1,
  "refresh": "30s",
  "time": {
    "from": "now-3h",
    "to": "now"
  },
  "editable": true,
  "templating": {
    "list": [
      {
        "name": "datasource",
        "label": "Datasource",
        "type": "datasource",
        "query": "prometheus",
        "current": {},
        "hide": 0
      },
      {
        "name": "namespace",
        "label": "Namespace",
        "type": "query",
        "datasource": {
          "type": "prometheus",
          "uid": "${datasource}"
        },
        "query": {
          "query": "label_values(container_cpu_usage_seconds_total{container!=\"\"}, namespace)",
          "refId": "ns"
        },
        "refresh": 2,
        "includeAll": true,
        "multi": true,
        "allValue": ".*",
        "current": {
          "text": "All",
          "value": "$__all"
        },
        "sort": 1
      },
      {
        "name": "pod",
        "label": "Pod",
        "type": "query",
        "datasource": {
          "type": "prometheus",
          "uid": "${datasource}"
        },
        "query": {
          "query": "label_values(container_cpu_usage_seconds_total{container!=\"\", namespace=~\"$namespace\"}, pod)",
          "refId": "pod"
        },
        "refresh": 2,
        "includeAll": true,
        "multi": true,
        "allValue": ".*",
        "current": {
          "text": "All",
          "value": "$__all"
        },
        "sort": 1
      }
    ]
  },
  "annotations": {
    "list": []
  },
  "panels": [
    {
      "id": 100,
      "type": "row",
      "title": "CFS quota throttling (cAdvisor)",
      "gridPos": {
        "x": 0,
        "y": 0,
        "w": 24,
        "h": 1
      },
      "collapsed": false,
      "panels": []
    },
    {
      "id": 1,
      "type": "timeseries",
      "title": "Throttle ratio per container",
      "description": "Share of 100ms CFS periods in which the container was throttled. > 5% on a latency-sensitive service is a red flag.",
      "datasource": {
        "type": "prometheus",
        "uid": "${datasource}"
      },
      "gridPos": {
        "x": 0,
        "y": 1,
        "w": 12,
        "h": 8
      },
      "fieldConfig": {
        "defaults": {
          "unit": "percentunit",
          "custom": {
            "lineWidth": 1,
            "fillOpacity": 10,
            "showPoints": "never",
            "thresholdsStyle": {
              "mode": "line"
            }
          },
          "thresholds": {
            "mode": "absolute",
            "steps": [
              {
                "color": "green",
                "value": null
              },
              {
                "color": "orange",
                "value": 0.05
              },
              {
                "color": "red",
                "value": 0.25
              }
            ]
          }
        },
        "overrides": []
      },
      "options": {
        "legend": {
          "displayMode": "table",
          "placement": "bottom",
          "calcs": [
            "mean",
            "max"
          ]
        },
        "tooltip": {
          "mode": "multi",
          "sort": "desc"
        }
      },
      "targets": [
        {
          "refId": "A",
          "datasource": {
            "type": "prometheus",
            "uid": "${datasource}"
          },
          "expr": "sum by (namespace, pod, container) (rate(container_cpu_cfs_throttled_periods_total{namespace=~\"$namespace\", pod=~\"$pod\", container!=\"\", container!=\"POD\"}[$__rate_interval]))\n/\nsum by (namespace, pod, container) (rate(container_cpu_cfs_periods_total{namespace=~\"$namespace\", pod=~\"$pod\", container!=\"\", container!=\"POD\"}[$__rate_interval]))",
          "legendFormat": "{{namespace}}/{{pod}}/{{container}}"
        }
      ]
    },
    {
      "id": 2,
      "type": "timeseries",
      "title": "Seconds frozen per second",
      "description": "Wall time the container's threads spent frozen by CFS quota, per second. Plot next to p99 latency: spikes line up.",
      "datasource": {
        "type": "prometheus",
        "uid": "${datasource}"
      },
      "gridPos": {
        "x": 12,
        "y": 1,
        "w": 12,
        "h": 8
      },
      "fieldConfig": {
        "defaults": {
          "unit": "s",
          "custom": {
            "lineWidth": 1,
            "fillOpacity": 10,
            "showPoints": "never"
          },
          "thresholds": {
            "mode": "absolute",
            "steps": [
              {
                "color": "green",
                "value": null
              }
            ]
          }
        },
        "overrides": []
      },
      "options": {
        "legend": {
          "displayMode": "table",
          "placement": "bottom",
          "calcs": [
            "mean",
            "max"
          ]
        },
        "tooltip": {
          "mode": "multi",
          "sort": "desc"
        }
      },
      "targets": [
        {
          "refId": "A",
          "datasource": {
            "type": "prometheus",
            "uid": "${datasource}"
          },
          "expr": "sum by (namespace, pod, container) (rate(container_cpu_cfs_throttled_seconds_total{namespace=~\"$namespace\", pod=~\"$pod\", container!=\"\", container!=\"POD\"}[$__rate_interval]))",
          "legendFormat": "{{namespace}}/{{pod}}/{{container}}"
        }
      ]
    },
    {
      "id": 3,
      "type": "timeseries",
      "title": "CPU usage vs limit",
      "description": "Used cores divided by CPU limit. Throttling while this is < 70% means bursty threads, not a lack of CPU.",
      "datasource": {
        "type": "prometheus",
        "uid": "${datasource}"
      },
      "gridPos": {
        "x": 0,
        "y": 9,
        "w": 12,
        "h": 8
      },
      "fieldConfig": {
        "defaults": {
          "unit": "percentunit",
          "custom": {
            "lineWidth": 1,
            "fillOpacity": 10,
            "showPoints": "never",
            "thresholdsStyle": {
              "mode": "line"
            }
          },
          "thresholds": {
            "mode": "absolute",
            "steps": [
              {
                "color": "green",
                "value": null
              },
              {
                "color": "orange",
                "value": 0.7
              },
              {
                "color": "red",
                "value": 0.9
              }
            ]
          }
        },
        "overrides": []
      },
      "options": {
        "legend": {
          "displayMode": "table",
          "placement": "bottom",
          "calcs": [
            "mean",
            "max"
          ]
        },
        "tooltip": {
          "mode": "multi",
          "sort": "desc"
        }
      },
      "targets": [
        {
          "refId": "A",
          "datasource": {
            "type": "prometheus",
            "uid": "${datasource}"
          },
          "expr": "sum by (namespace, pod, container) (rate(container_cpu_usage_seconds_total{namespace=~\"$namespace\", pod=~\"$pod\", container!=\"\", container!=\"POD\"}[$__rate_interval]))\n/\nsum by (namespace, pod, container) (kube_pod_container_resource_limits{resource=\"cpu\", namespace=~\"$namespace\", pod=~\"$pod\"})",
          "legendFormat": "{{namespace}}/{{pod}}/{{container}}"
        }
      ]
    },
    {
      "id": 4,
      "type": "timeseries",
      "title": "CPU usage vs request",
      "description": "Used cores divided by CPU request. Sustained > 100% means requests are too low for scheduling.",
      "datasource": {
        "type": "prometheus",
        "uid": "${datasource}"
      },
      "gridPos": {
        "x": 12,
        "y": 9,
        "w": 12,
        "h": 8
      },
      "fieldConfig": {
        "defaults": {
          "unit": "percentunit",
          "custom": {
            "lineWidth": 1,
            "fillOpacity": 10,
            "showPoints": "never",
            "thresholdsStyle": {
              "mode": "line"
            }
          },
          "thresholds": {
            "mode": "absolute",
            "steps": [
              {
                "color": "green",
                "value": null
              },
              {
                "color": "orange",
                "value": 1
              }
            ]
          }
        },
        "overrides": []
      },
      "options": {
        "legend": {
          "displayMode": "table",
          "placement": "bottom",
          "calcs": [
            "mean",
            "max"
          ]
        },
        "tooltip": {
          "mode": "multi",
          "sort": "desc"
        }
      },
      "targets": [
        {
          "refId": "A",
          "datasource": {
            "type": "prometheus",
            "uid": "${datasource}"
          },
          "expr": "sum by (namespace, pod, container) (rate(container_cpu_usage_seconds_total{namespace=~\"$namespace\", pod=~\"$pod\", container!=\"\", container!=\"POD\"}[$__rate_interval]))\n/\nsum by (namespace, pod, container) (kube_pod_container_resource_requests{resource=\"cpu\", namespace=~\"$namespace\", pod=~\"$pod\"})",
          "legendFormat": "{{namespace}}/{{pod}}/{{container}}"
        }
      ]
    },
    {
      "id": 5,
      "type": "table",
      "title": "Top 15 throttled containers (current)",
      "datasource": {
        "type": "prometheus",
        "uid": "${datasource}"
      },
      "gridPos": {
        "x": 0,
        "y": 17,
        "w": 24,
        "h": 8
      },
      "description": "Throttle ratio alongside usage/limit. High throttle + low usage = reduce thread count or raise the limit.",
      "fieldConfig": {
        "defaults": {
          "unit": "percentunit",
          "decimals": 1
        },
        "overrides": []
      },
      "options": {
        "showHeader": true,
        "sortBy": [
          {
            "displayName": "Throttle ratio",
            "desc": true
          }
        ]
      },
      "targets": [
        {
          "refId": "A",
          "datasource": {
            "type": "prometheus",
            "uid": "${datasource}"
          },
          "instant": true,
          "format": "table",
          "expr": "topk(15, sum by (namespace, pod, container) (rate(container_cpu_cfs_throttled_periods_total{namespace=~\"$namespace\", pod=~\"$pod\", container!=\"\", container!=\"POD\"}[5m])) / sum by (namespace, pod, container) (rate(container_cpu_cfs_periods_total{namespace=~\"$namespace\", pod=~\"$pod\", container!=\"\", container!=\"POD\"}[5m])))"
        },
        {
          "refId": "B",
          "datasource": {
            "type": "prometheus",
            "uid": "${datasource}"
          },
          "instant": true,
          "format": "table",
          "expr": "sum by (namespace, pod, container) (rate(container_cpu_usage_seconds_total{namespace=~\"$namespace\", pod=~\"$pod\", container!=\"\", container!=\"POD\"}[5m])) / sum by (namespace, pod, container) (kube_pod_container_resource_limits{resource=\"cpu\", namespace=~\"$namespace\", pod=~\"$pod\"})"
        }
      ],
      "transformations": [
        {
          "id": "merge",
          "options": {}
        },
        {
          "id": "organize",
          "options": {
            "excludeByName": {
              "Time": true
            },
            "renameByName": {
              "Value #A": "Throttle ratio",
              "Value #B": "Usage / limit"
            }
          }
        },
        {
          "id": "filterByValue",
          "options": {
            "type": "include",
            "match": "all",
            "filters": [
              {
                "fieldName": "Throttle ratio",
                "config": {
                  "id": "greater",
                  "options": {
                    "value": 0
                  }
                }
              }
            ]
          }
        }
      ]
    },
    {
      "id": 101,
      "type": "row",
      "title": "CPU pressure and waiting (PSI, steal, scheduler latency)",
      "gridPos": {
        "x": 0,
        "y": 25,
        "w": 24,
        "h": 1
      },
      "collapsed": false,
      "panels": []
    },
    {
      "id": 6,
      "type": "timeseries",
      "title": "Container CPU pressure (PSI)",
      "description": "Fraction of time at least one task in the container waited for CPU. Needs cgroup v2 + kubelet/cAdvisor with PSI enabled.",
      "datasource": {
        "type": "prometheus",
        "uid": "${datasource}"
      },
      "gridPos": {
        "x": 0,
        "y": 26,
        "w": 12,
        "h": 8
      },
      "fieldConfig": {
        "defaults": {
          "unit": "percentunit",
          "custom": {
            "lineWidth": 1,
            "fillOpacity": 10,
            "showPoints": "never",
            "thresholdsStyle": {
              "mode": "line"
            }
          },
          "thresholds": {
            "mode": "absolute",
            "steps": [
              {
                "color": "green",
                "value": null
              },
              {
                "color": "orange",
                "value": 0.1
              },
              {
                "color": "red",
                "value": 0.3
              }
            ]
          }
        },
        "overrides": []
      },
      "options": {
        "legend": {
          "displayMode": "table",
          "placement": "bottom",
          "calcs": [
            "mean",
            "max"
          ]
        },
        "tooltip": {
          "mode": "multi",
          "sort": "desc"
        }
      },
      "targets": [
        {
          "refId": "A",
          "datasource": {
            "type": "prometheus",
            "uid": "${datasource}"
          },
          "expr": "sum by (namespace, pod, container) (rate(container_pressure_cpu_waiting_seconds_total{namespace=~\"$namespace\", pod=~\"$pod\", container!=\"\", container!=\"POD\"}[$__rate_interval]))",
          "legendFormat": "{{namespace}}/{{pod}}/{{container}}"
        }
      ]
    },
    {
      "id": 7,
      "type": "timeseries",
      "title": "Node CPU pressure (PSI) and steal",
      "description": "Node-level waiting (node-exporter PSI) and hypervisor steal. Steal > 2-10% sustained = noisy neighbour.",
      "datasource": {
        "type": "prometheus",
        "uid": "${datasource}"
      },
      "gridPos": {
        "x": 12,
        "y": 26,
        "w": 12,
        "h": 8
      },
      "fieldConfig": {
        "defaults": {
          "unit": "percentunit",
          "custom": {
            "lineWidth": 1,
            "fillOpacity": 10,
            "showPoints": "never",
            "thresholdsStyle": {
              "mode": "line"
            }
          },
          "thresholds": {
            "mode": "absolute",
            "steps": [
              {
                "color": "green",
                "value": null
              },
              {
                "color": "orange",
                "value": 0.05
              }
            ]
          }
        },
        "overrides": []
      },
      "options": {
        "legend": {
          "displayMode": "table",
          "placement": "bottom",
          "calcs": [
            "mean",
            "max"
          ]
        },
        "tooltip": {
          "mode": "multi",
          "sort": "desc"
        }
      },
      "targets": [
        {
          "refId": "A",
          "datasource": {
            "type": "prometheus",
            "uid": "${datasource}"
          },
          "expr": "rate(node_pressure_cpu_waiting_seconds_total[$__rate_interval])",
          "legendFormat": "psi {{instance}}"
        },
        {
          "refId": "B",
          "datasource": {
            "type": "prometheus",
            "uid": "${datasource}"
          },
          "expr": "sum by (instance) (rate(node_cpu_seconds_total{mode=\"steal\"}[$__rate_interval]))\n/\ncount by (instance) (node_cpu_seconds_total{mode=\"idle\"})",
          "legendFormat": "steal {{instance}}"
        }
      ]
    },
    {
      "id": 8,
      "type": "timeseries",
      "title": "Go scheduler latency p99 / p50",
      "description": "Time goroutines waited in the run queue (runtime/metrics /sched/latencies:seconds). Enable with client_golang WithGoCollectorRuntimeMetrics. p99 in ms = CPU starvation.",
      "datasource": {
        "type": "prometheus",
        "uid": "${datasource}"
      },
      "gridPos": {
        "x": 0,
        "y": 34,
        "w": 24,
        "h": 8
      },
      "fieldConfig": {
        "defaults": {
          "unit": "s",
          "custom": {
            "lineWidth": 1,
            "fillOpacity": 10,
            "showPoints": "never",
            "thresholdsStyle": {
              "mode": "line"
            }
          },
          "thresholds": {
            "mode": "absolute",
            "steps": [
              {
                "color": "green",
                "value": null
              },
              {
                "color": "orange",
                "value": 0.001
              },
              {
                "color": "red",
                "value": 0.01
              }
            ]
          }
        },
        "overrides": []
      },
      "options": {
        "legend": {
          "displayMode": "table",
          "placement": "bottom",
          "calcs": [
            "mean",
            "max"
          ]
        },
        "tooltip": {
          "mode": "multi",
          "sort": "desc"
        }
      },
      "targets": [
        {
          "refId": "A",
          "datasource": {
            "type": "prometheus",
            "uid": "${datasource}"
          },
          "expr": "histogram_quantile(0.99, sum by (le, pod) (rate(go_sched_latencies_seconds_bucket{namespace=~\"$namespace\", pod=~\"$pod\"}[$__rate_interval])))",
          "legendFormat": "p99 {{pod}}"
        },
        {
          "refId": "B",
          "datasource": {
            "type": "prometheus",
            "uid": "${datasource}"
          },
          "expr": "histogram_quantile(0.50, sum by (le, pod) (rate(go_sched_latencies_seconds_bucket{namespace=~\"$namespace\", pod=~\"$pod\"}[$__rate_interval])))",
          "legendFormat": "p50 {{pod}}"
        }
      ]
    }
  ]
}