Files
sshclient/core/monitor.py
T
Hermes 5e69b1c3e2 feat: overhaul process monitoring UI - more info, tree view, details, batch ops
User feedback: '进程比较关键,以下显示太少了,需要继续优化下监控UI'

Process monitor panel was rebuilt:

1. core/monitor.py: 200 -> 500 processes, added PPID / PRI / NICE
   New ps fields: pid ppid user pcpu pmem vsz rss stat pri nice etimes times args
   Python parser: new fields ppid, pri, nice

2. ui/widgets.py MonitorPanel:
   a) Toolbar: search box + sort dropdown (CPU/MEM/PID/start/user/comm)
      + view toggle (Flat / Tree) + CPU% min filter (SpinBox 0-100%)
      + batch kill button
   b) Table: 12 columns (Flat) / 13 cols (Tree + indent col)
      PID PPID USER CPU% MEM% RSS STAT PRI NI '启动' 'CPU时间' 命令
      - Compact row height 20px (was default 30)
      - Extended selection (Ctrl/Shift multi)
      - Color highlights:
          CPU% >=80 red bg+fg bold, >=50 red fg bold, >=20 orange
          MEM% >=15 orange bg+fg bold, >=5 orange
          STAT: Z orange bg + orange-red fg bold, R green, D purple
          NI: negative green, positive orange
      - Red bg on EXTREME rows for instant visibility
   c) Tree view: builds PPID tree with ├─/└─/│ connectors;
      root processes (PID 1) at depth 0, sorted by CPU within siblings.
      Useful for 'which process is spawning these things?'
   d) Right side panel:
      - 进程详情 (QTextEdit dark theme): full PID/PPID/USER/STAT/PRI/NI/
        CPU/MEM/RSS/VSZ/etime/cputime, command, parent process name,
        child processes list (top 10)
      - 🔥 CPU TOP 5 (QListWidget dark): click to jump+select in main table
      - 💾 内存 TOP 5 (QListWidget dark): same
   e) Right-click menu enhanced:
      kill (SIGTERM) | kill -9 (SIGKILL) | batch kill (if multi-select)
      | copy PID | copy command | filter by command | view details
   f) Batch kill: collects all selected PIDs, runs 'kill p1 p2 p3 ...'
      in a single SSH call, then loops 'kill -0' to report survivors.
      One confirmation dialog lists top 10 + '还有 N 个'.
   g) Status row at bottom: 'Ctrl+A 全选, Shift+点击多选, 右键批量操作'

3. test_process_monitor.py: 5/5 tests pass with 500 rows, tree view,
   TOP 5, CPU min filter, etc.

All tests pass: core 6/6 + UI 3/3 + E2E 6/6 + terminal 5/5 + proc 5/5
Build: 57 MB single-file exe.
2026-07-28 22:39:26 +08:00

226 lines
9.0 KiB
Python

"""
远程主机系统监控模块
通过 SSH 一次性采集 CPU/内存/磁盘/网络/负载指标。
Linux 用 /proc 和常用命令;macOS/BSD 走兼容路径。
"""
import re
import time
from typing import Optional
from .ssh_client import SSHConnection
class SystemMonitor:
"""远程主机的资源监控器(数据全部从 SSH 通道采集,不依赖 agent)"""
# 一次性获取所有指标的脚本(Linux)
_LINUX_METRICS_SCRIPT = r"""
echo "===CPU==="
# 第一次采样 1 秒间隔,用来计算差值
read cpu_user cpu_nice cpu_system cpu_idle cpu_iowait cpu_irq cpu_softirq cpu_steal < /proc/stat
sleep 1
read cpu_user2 cpu_nice2 cpu_system2 cpu_idle2 cpu_iowait2 cpu_irq2 cpu_softirq2 cpu_steal2 < /proc/stat
total1=$((cpu_user+cpu_nice+cpu_system+cpu_idle+cpu_iowait+cpu_irq+cpu_softirq+cpu_steal))
total2=$((cpu_user2+cpu_nice2+cpu_system2+cpu_idle2+cpu_iowait2+cpu_irq2+cpu_softirq2+cpu_steal2))
idle1=$cpu_idle; idle2=$cpu_idle2
dt=$((total2-total1)); di=$((idle2-idle1))
if [ $dt -gt 0 ]; then usage=$(( (1000*(dt-di)/dt+5)/10 )); else usage=0; fi
echo "CPU_USAGE=$usage"
echo "CPU_CORES=$(nproc 2>/dev/null || echo 1)"
echo "LOAD=$(cat /proc/loadavg | awk '{print $1,$2,$3}')"
echo "UPTIME=$(awk '{printf "%.0f",$1}' /proc/uptime)"
echo "===MEM==="
mem_total=$(awk '/MemTotal/{print $2}' /proc/meminfo)
mem_avail=$(awk '/MemAvailable/{print $2}' /proc/meminfo)
swap_total=$(awk '/SwapTotal/{print $2}' /proc/meminfo)
swap_free=$(awk '/SwapFree/{print $2}' /proc/meminfo)
if [ -z "$mem_avail" ]; then mem_avail=$((mem_total - $(awk '/^(Buffers|Cached|SReclaimable):/{s+=$2} END{print s}' /proc/meminfo))); fi
used=$((mem_total - mem_avail))
echo "MEM_TOTAL=$mem_total"
echo "MEM_USED=$used"
echo "MEM_AVAIL=$mem_avail"
echo "SWAP_TOTAL=$swap_total"
echo "SWAP_USED=$((swap_total-swap_free))"
echo "===DISK==="
df -PB1 -x tmpfs -x devtmpfs 2>/dev/null | awk 'NR>1 {printf "DISK|%s|%d|%d|%s\n",$NF,$2,$3,$5}'
echo "===NET==="
for iface in $(ls /sys/class/net/ 2>/dev/null | grep -v lo); do
rx=$(cat /sys/class/net/$iface/statistics/rx_bytes 2>/dev/null || echo 0)
tx=$(cat /sys/class/net/$iface/statistics/tx_bytes 2>/dev/null || echo 0)
echo "NET|$iface|$rx|$tx"
done
echo "===PROC==="
# 进程统计
ps -e -o stat= 2>/dev/null | awk '{r+=($1~/^R/); s+=($1~/^S/); d+=($1~/^D/); z+=($1~/^Z/); t++} END{printf "PROC_TOTAL=%d\nPROC_RUNNING=%d\nPROC_SLEEP=%d\nPROC_DISK=%d\nPROC_ZOMBIE=%d\n", t, r, s, d, z}'
# 进程列表:etimes/times 都是纯数字,避免 start 字段含 "Jul 21" 多列错位
# comm 字段可能含空格,前 9 列用 \t 拼,comm 用换行做记录结束
# 字段:pid ppid user pcpu pmem vsz rss stat pri nice etimes times args
ps -eo pid,ppid,user,pcpu,pmem,vsz,rss,stat,pri,nice,etimes,times,args --sort=-pcpu --no-headers 2>/dev/null \
| head -500 \
| awk '{
out="PROC\t";
for(i=1;i<=12;i++) out=out $i "\t";
rest="";
for(i=13;i<=NF;i++) rest=(i==13?$i:rest " " $i);
print out rest
}'
echo "===HOST==="
echo "HOSTNAME=$(hostname)"
echo "KERNEL=$(uname -r)"
echo "OS=$(. /etc/os-release 2>/dev/null && echo "$PRETTY_NAME" || uname -s)"
"""
@staticmethod
def _parse_kv(text: str, key: str, default: str = "0") -> str:
"""从 KEY=VALUE 行中取值"""
m = re.search(rf"^{re.escape(key)}=(.+)$", text, re.MULTILINE)
return m.group(1).strip() if m else default
@classmethod
def collect(cls, conn: SSHConnection) -> dict:
"""采集一次指标;返回 dict"""
empty = {
"cpu": 0.0, "cores": 1, "load1": 0, "load5": 0, "load15": 0,
"uptime": 0, "hostname": "", "kernel": "", "os": "",
"mem_total": 0, "mem_used": 0, "mem_percent": 0.0,
"swap_total": 0, "swap_used": 0,
"disks": [], "net": [],
"proc_total": 0, "proc_running": 0, "proc_sleep": 0,
"proc_disk": 0, "proc_zombie": 0, "processes": [],
"ts": time.time(),
}
if not conn or not conn.connected:
return empty
code, out, err = conn.exec_command(cls._LINUX_METRICS_SCRIPT, timeout=10)
if code != 0 or not out:
empty["error"] = err or "采集失败"
return empty
result = dict(empty)
result["hostname"] = cls._parse_kv(out, "HOSTNAME")
result["kernel"] = cls._parse_kv(out, "KERNEL")
result["os"] = cls._parse_kv(out, "OS")
try:
result["cpu"] = float(cls._parse_kv(out, "CPU_USAGE"))
except ValueError:
pass
try:
result["cores"] = int(cls._parse_kv(out, "CPU_CORES", "1"))
except ValueError:
pass
load = cls._parse_kv(out, "LOAD", "0 0 0").split()
try:
result["load1"] = float(load[0])
result["load5"] = float(load[1]) if len(load) > 1 else 0
result["load15"] = float(load[2]) if len(load) > 2 else 0
except (ValueError, IndexError):
pass
try:
result["uptime"] = int(cls._parse_kv(out, "UPTIME"))
except ValueError:
pass
try:
mt = int(cls._parse_kv(out, "MEM_TOTAL"))
mu = int(cls._parse_kv(out, "MEM_USED"))
result["mem_total"] = mt
result["mem_used"] = mu
result["mem_percent"] = (mu / mt * 100) if mt > 0 else 0.0
result["swap_total"] = int(cls._parse_kv(out, "SWAP_TOTAL"))
result["swap_used"] = int(cls._parse_kv(out, "SWAP_USED"))
except ValueError:
pass
result["disks"] = []
for line in out.splitlines():
if line.startswith("DISK|"):
_, mount, total, used, percent = line.split("|", 4)
try:
result["disks"].append({
"mount": mount, "total": int(total),
"used": int(used), "percent": int(percent.rstrip("%")),
})
except ValueError:
continue
result["net"] = []
for line in out.splitlines():
if line.startswith("NET|"):
_, name, rx, tx = line.split("|", 3)
try:
result["net"].append({
"iface": name, "rx": int(rx), "tx": int(tx),
})
except ValueError:
continue
# 进程统计
try:
result["proc_total"] = int(cls._parse_kv(out, "PROC_TOTAL"))
result["proc_running"] = int(cls._parse_kv(out, "PROC_RUNNING"))
result["proc_sleep"] = int(cls._parse_kv(out, "PROC_SLEEP"))
result["proc_disk"] = int(cls._parse_kv(out, "PROC_DISK"))
result["proc_zombie"] = int(cls._parse_kv(out, "PROC_ZOMBIE"))
except ValueError:
result["proc_total"] = 0
result["proc_running"] = 0
result["proc_sleep"] = 0
result["proc_disk"] = 0
result["proc_zombie"] = 0
# 进程列表(远程用 \t 分隔前 12 列,第 13 列开始是 args/comm,可能含空格)
result["processes"] = []
for line in out.splitlines():
if not line.startswith("PROC\t"):
continue
payload = line[5:] # 去掉 "PROC\t" 前缀
fields = payload.split("\t", 12) # 只切前 12 次
if len(fields) < 13:
continue
try:
result["processes"].append({
"pid": int(fields[0]),
"ppid": int(fields[1]) if fields[1].isdigit() else 0,
"user": fields[2],
"pcpu": float(fields[3]),
"pmem": float(fields[4]),
"vsz": int(fields[5]),
"rss": int(fields[6]),
"stat": fields[7],
"pri": int(fields[8]) if fields[8].lstrip("-").isdigit() else 0,
"nice": int(fields[9]) if fields[9].lstrip("-").isdigit() else 0,
# etimes: 自启动以来的秒数(整数)
"etime": int(fields[10]) if fields[10].isdigit() else 0,
# times: 累计 CPU 时间秒数
"time": int(fields[11]) if fields[11].isdigit() else 0,
"comm": fields[12].strip(),
})
except (ValueError, IndexError):
continue
return result
@staticmethod
def format_bytes(n: int) -> str:
"""人类可读字节数"""
n = float(n)
for unit in ("B", "KB", "MB", "GB", "TB", "PB"):
if n < 1024:
return f"{n:.1f}{unit}"
n /= 1024
return f"{n:.1f}EB"
@staticmethod
def format_uptime(seconds: int) -> str:
seconds = int(seconds)
d, rem = divmod(seconds, 86400)
h, rem = divmod(rem, 3600)
m, s = divmod(rem, 60)
if d:
return f"{d}{h}小时"
if h:
return f"{h}小时{m}"
if m:
return f"{m}{s}"
return f"{s}"