feat: k8s-status 集群巡检技能(SKILL.md + 脚本)

This commit is contained in:
NightStar
2026-08-04 20:51:57 +08:00
parent 0ccb17d535
commit 5ce3695563
2 changed files with 96 additions and 0 deletions
+39
View File
@@ -0,0 +1,39 @@
---
name: "k8s-status"
description: "巡检 K3s/K8s 集群:kubectl 读取节点状态、资源占用、Pod 分布、异常项,解析输出核心字段。"
---
# K8s/K3s 集群状态巡检
通过 kubectl 读取多节点集群运行状态,解析核心字段并输出结构化报告。
适用场景:用户问「集群/节点/服务现在什么状态」「帮我看看机器跑得怎么样」。
## 用法
```bash
# vps1(control-plane,kubectl 直接可用)
bash scripts/k8s-status.sh
# 指定 kubeconfig(从任意机器远程跑)
KUBECONFIG=/etc/rancher/k3s/k3s.yaml bash scripts/k8s-status.sh
```
环境变量:
- `KUBECTL` — kubectl 路径,默认 `kubectl`
- `KUBECONFIG` — kubeconfig 路径,默认当前上下文
## 输出内容
1. **节点状态全量** — `kubectl get nodes -o wide`
2. **节点健康解析** — jsonpath 精确提取:名称/Ready状态/角色(control-plane|worker)/版本/内网IP/公网IP/OS/内核/运行时
3. **资源占用** — `kubectl top nodes`(metrics-server 不可用时自动跳过)
4. **Pod 总数 & 节点分布**
5. **异常 Pod** — 非 Running/Completed 状态列表
6. **Warning 事件** — 最近 5 条
## 注意事项
- 用 jsonpath + tab 分隔提取字段,避免 `-o wide` 空格切分错位(OS-IMAGE 含空格会坑 awk)
- 角色判定:`node-role.kubernetes.io/control-plane` label 为 true → control-plane,否则 worker
- metrics-server 对部分节点返回 `<unknown>`:通常是跨节点网络问题(如 flannel VXLAN 不通),不是脚本 bug
- Warning 事件 `InvalidDiskCapacity`(invalid capacity 0 on image filesystem)常见于容器内 kubelet 或磁盘识别异常,需单独排查
+57
View File
@@ -0,0 +1,57 @@
#!/usr/bin/env bash
# k8s-status.sh — 通过 kubectl 读取 K3s 集群状态,解析输出核心字段
# 用法:
# bash k8s-status.sh # 默认 kubectl + 当前 kubeconfig
# KUBECONFIG=/etc/rancher/k3s/k3s.yaml bash k8s-status.sh
# 环境变量: KUBECTL(kubectl 路径,默认 kubectl)、KUBECONFIG
set -euo pipefail
KUBECTL_CMD=(kubectl)
if [[ -n "${KUBECTL:-}" ]]; then
KUBECTL_CMD=("$KUBECTL")
fi
if [[ -n "${KUBECONFIG:-}" ]]; then
KUBECTL_CMD+=(--kubeconfig "$KUBECONFIG")
fi
echo "===================== K3s 集群状态巡检 ====================="
echo "时间: $(date '+%Y-%m-%d %H:%M:%S %Z')"
echo
echo "────── ① 节点状态(全量 -o wide)──────"
"${KUBECTL_CMD[@]}" get nodes -o wide
echo
echo "────── ② 节点健康解析(核心字段,jsonpath 精确提取)──────"
"${KUBECTL_CMD[@]}" get nodes -o jsonpath='
{range .items[*]}
{ .metadata.name}{"\t"}{ .status.conditions[?(@.type=="Ready")].status}{"\t"}{ .metadata.labels.node-role\.kubernetes\.io/control-plane}{"\t"}{ .status.nodeInfo.kubeletVersion}{"\t"}{ .status.addresses[?(@.type=="InternalIP")].address}{"\t"}{ .status.addresses[?(@.type=="ExternalIP")].address}{"\t"}{ .status.nodeInfo.osImage}{"\t"}{ .status.nodeInfo.kernelVersion}{"\t"}{ .status.nodeInfo.containerRuntimeVersion}{"\n"}
{end}' | awk -F'\t' 'NF {
role = ($3 == "true") ? "control-plane" : "worker"
printf " %-26s %-7s %-14s %-14s 内网IP=%-16s 公网IP=%-16s %s | %s | %s\n", $1, $2, role, $4, $5, $6, $7, $8, $9
}'
echo
echo "────── ③ 节点资源占用(CPU/内存)──────"
if "${KUBECTL_CMD[@]}" top nodes 2>/dev/null; then
:
else
echo " (metrics-server 暂不可用,跳过)"
fi
echo
echo "────── ④ Pod 总数 & 按节点分布 ──────"
TOTAL=$("${KUBECTL_CMD[@]}" get pods -A --no-headers 2>/dev/null | wc -l | tr -d ' ')
echo " 集群 Pod 总数: ${TOTAL:-0}"
"${KUBECTL_CMD[@]}" get pods -A -o wide --no-headers 2>/dev/null | awk '{print $8}' | sort | uniq -c | awk '{printf " %s: %s 个 pod\n", $2, $1}'
echo
echo "────── ⑤ 异常 Pod(非 Running/Completed)──────"
"${KUBECTL_CMD[@]}" get pods -A --no-headers 2>/dev/null | awk '$4 != "Running" && $4 != "Completed" { printf " [%s] %s/%s: %s (%s)\n", $1, $2, $3, $4, $5 }'
echo " (无输出即全部健康)"
echo
echo "────── ⑥ 最近 Warning 事件(最多 5 条)──────"
"${KUBECTL_CMD[@]}" get events -A --field-selector type=Warning --sort-by=.lastTimestamp 2>/dev/null | awk 'NR==1 {next} {print " " $0}' | tail -5 || true
echo
echo "===================== 巡检完成 ====================="