1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
| #!/usr/bin/env python3
# scripts/fault_diagnosis.py
import os
import subprocess
import re
import json
import time
import psutil
from datetime import datetime
from typing import Dict, List, Tuple
class ServerDiagnostic:
def __init__(self):
self.results = {}
self.thresholds = {
'cpu_usage': 80.0,
'memory_usage': 85.0,
'disk_usage': 90.0,
'load_average': 2.0,
'response_time': 5.0
}
def diagnose_system(self) -> Dict:
"""执行完整的系统诊断"""
self.results['timestamp'] = datetime.now().isoformat()
self.results['hostname'] = os.uname().nodename
# 基础系统检查
self.results['cpu'] = self.check_cpu()
self.results['memory'] = self.check_memory()
self.results['disk'] = self.check_disk()
self.results['network'] = self.check_network()
self.results['processes'] = self.check_processes()
self.results['services'] = self.check_services()
self.results['logs'] = self.check_logs()
# 性能检查
self.results['performance'] = self.check_performance()
# 安全检查
self.results['security'] = self.check_security()
# 生成诊断报告
self.results['summary'] = self.generate_summary()
return self.results
def check_cpu(self) -> Dict:
"""检查CPU状态"""
try:
cpu_percent = psutil.cpu_percent(interval=1)
load_avg = os.getloadavg()
cpu_count = psutil.cpu_count()
cpu_info = {
'usage_percent': cpu_percent,
'load_average': {
'1min': load_avg[0],
'5min': load_avg[1],
'15min': load_avg[2]
},
'cpu_count': cpu_count,
'per_cpu_usage': psutil.cpu_percent(percpu=True)
}
# 检查是否有问题
cpu_info['issues'] = []
if cpu_percent > self.thresholds['cpu_usage']:
cpu_info['issues'].append(f"High CPU usage: {cpu_percent}%")
if load_avg[0] > self.thresholds['load_average'] * cpu_count:
cpu_info['issues'].append(f"High load average: {load_avg[0]}")
return cpu_info
except Exception as e:
return {'error': str(e)}
def check_memory(self) -> Dict:
"""检查内存状态"""
try:
memory = psutil.virtual_memory()
swap = psutil.swap_memory()
memory_info = {
'total': memory.total,
'available': memory.available,
'used': memory.used,
'free': memory.free,
'usage_percent': memory.percent,
'swap': {
'total': swap.total,
'used': swap.used,
'free': swap.free,
'usage_percent': swap.percent
}
}
memory_info['issues'] = []
if memory.percent > self.thresholds['memory_usage']:
memory_info['issues'].append(f"High memory usage: {memory.percent}%")
if swap.percent > 50:
memory_info['issues'].append(f"High swap usage: {swap.percent}%")
return memory_info
except Exception as e:
return {'error': str(e)}
def check_disk(self) -> Dict:
"""检查磁盘状态"""
try:
disk_partitions = psutil.disk_partitions()
disk_info = {'partitions': [], 'issues': []}
for partition in disk_partitions:
try:
usage = psutil.disk_usage(partition.mountpoint)
partition_info = {
'device': partition.device,
'mountpoint': partition.mountpoint,
'fstype': partition.fstype,
'total': usage.total,
'used': usage.used,
'free': usage.free,
'usage_percent': (usage.used / usage.total) * 100
}
if partition_info['usage_percent'] > self.thresholds['disk_usage']:
disk_info['issues'].append(
f"Low disk space on {partition.mountpoint}: {partition_info['usage_percent']:.1f}%"
)
disk_info['partitions'].append(partition_info)
except PermissionError:
continue
return disk_info
except Exception as e:
return {'error': str(e)}
def check_network(self) -> Dict:
"""检查网络状态"""
try:
network_info = {
'interfaces': [],
'connections': [],
'issues': []
}
# 检查网络接口
net_io = psutil.net_io_counters(pernic=True)
for interface, stats in net_io.items():
interface_info = {
'name': interface,
'bytes_sent': stats.bytes_sent,
'bytes_recv': stats.bytes_recv,
'packets_sent': stats.packets_sent,
'packets_recv': stats.packets_recv,
'errors_in': stats.errin,
'errors_out': stats.errout,
'drop_in': stats.dropin,
'drop_out': stats.dropout
}
network_info['interfaces'].append(interface_info)
# 检查网络连接
connections = psutil.net_connections()
connection_stats = {
'established': 0,
'listening': 0,
'time_wait': 0,
'total': len(connections)
}
for conn in connections:
if conn.status == 'ESTABLISHED':
connection_stats['established'] += 1
elif conn.status == 'LISTEN':
connection_stats['listening'] += 1
elif conn.status == 'TIME_WAIT':
connection_stats['time_wait'] += 1
network_info['connection_stats'] = connection_stats
# 检查是否有异常
if connection_stats['time_wait'] > 1000:
network_info['issues'].append(f"Too many TIME_WAIT connections: {connection_stats['time_wait']}")
return network_info
except Exception as e:
return {'error': str(e)}
def check_processes(self) -> Dict:
"""检查进程状态"""
try:
processes = []
top_cpu = []
top_memory = []
for proc in psutil.process_iter(['pid', 'name', 'cpu_percent', 'memory_percent', 'status']):
try:
proc_info = proc.info
processes.append(proc_info)
# 记录CPU使用率最高的进程
if len(top_cpu) < 10:
top_cpu.append(proc_info)
else:
min_proc = min(top_cpu, key=lambda x: x['cpu_percent'])
if proc_info['cpu_percent'] > min_proc['cpu_percent']:
top_cpu.remove(min_proc)
top_cpu.append(proc_info)
# 记录内存使用率最高的进程
if len(top_memory) < 10:
top_memory.append(proc_info)
else:
min_proc = min(top_memory, key=lambda x: x['memory_percent'])
if proc_info['memory_percent'] > min_proc['memory_percent']:
top_memory.remove(min_proc)
top_memory.append(proc_info)
except (psutil.NoSuchProcess, psutil.AccessDenied):
continue
return {
'total_processes': len(processes),
'top_cpu_processes': sorted(top_cpu, key=lambda x: x['cpu_percent'], reverse=True),
'top_memory_processes': sorted(top_memory, key=lambda x: x['memory_percent'], reverse=True)
}
except Exception as e:
return {'error': str(e)}
def check_services(self) -> Dict:
"""检查关键服务状态"""
critical_services = ['nginx', 'mysql', 'redis-server', 'ssh']
service_status = {}
for service in critical_services:
try:
# 检查服务是否运行
result = subprocess.run(
['systemctl', 'is-active', service],
capture_output=True, text=True
)
service_status[service] = {
'status': result.stdout.strip(),
'active': result.stdout.strip() == 'active'
}
except Exception as e:
service_status[service] = {
'status': 'unknown',
'active': False,
'error': str(e)
}
return service_status
def check_logs(self) -> Dict:
"""检查系统日志中的错误"""
log_files = ['/var/log/syslog', '/var/log/auth.log', '/var/log/nginx/error.log']
log_errors = []
for log_file in log_files:
if os.path.exists(log_file):
try:
# 读取最近的错误日志
result = subprocess.run(
['tail', '-100', log_file],
capture_output=True, text=True
)
errors = re.findall(r'.*error.*|.*ERROR.*|.*Error.*', result.stdout, re.IGNORECASE)
if errors:
log_errors.extend([
{'file': log_file, 'error': error.strip()}
for error in errors[-10:] # 最近10个错误
])
except Exception as e:
log_errors.append({'file': log_file, 'error': f'Failed to read log: {str(e)}'})
return {'errors': log_errors}
def generate_summary(self) -> Dict:
"""生成诊断摘要"""
issues = []
# 收集所有问题
if 'cpu' in self.results and 'issues' in self.results['cpu']:
issues.extend(self.results['cpu']['issues'])
if 'memory' in self.results and 'issues' in self.results['memory']:
issues.extend(self.results['memory']['issues'])
if 'disk' in self.results and 'issues' in self.results['disk']:
issues.extend(self.results['disk']['issues'])
if 'network' in self.results and 'issues' in self.results['network']:
issues.extend(self.results['network']['issues'])
# 检查服务状态
if 'services' in self.results:
for service, status in self.results['services'].items():
if not status.get('active', False):
issues.append(f"Service {service} is not running")
return {
'total_issues': len(issues),
'issues': issues,
'health_score': max(0, 100 - len(issues) * 10),
'recommendations': self.generate_recommendations(issues)
}
def generate_recommendations(self, issues: List[str]) -> List[str]:
"""根据问题生成建议"""
recommendations = []
for issue in issues:
if 'CPU' in issue:
recommendations.append("Consider upgrading CPU or optimizing CPU-intensive processes")
elif 'memory' in issue.lower():
recommendations.append("Add more RAM or optimize memory usage")
elif 'disk' in issue.lower():
recommendations.append("Clean up disk space or add more storage")
elif 'service' in issue.lower():
recommendations.append("Restart failed services or check service configuration")
elif 'network' in issue.lower():
recommendations.append("Check network configuration and bandwidth")
return list(set(recommendations)) # 去重
def main():
"""主函数"""
diagnostic = ServerDiagnostic()
results = diagnostic.diagnose_system()
# 保存诊断结果
timestamp = datetime.now().strftime('%Y%m%d_%H%M%S')
output_file = f"/tmp/server_diagnostic_{timestamp}.json"
with open(output_file, 'w') as f:
json.dump(results, f, indent=2, default=str)
print(f"Diagnostic completed. Results saved to: {output_file}")
# 如果有问题,发送告警
if results['summary']['total_issues'] > 0:
print(f"Found {results['summary']['total_issues']} issues:")
for issue in results['summary']['issues']:
print(f" - {issue}")
if __name__ == '__main__':
main()
|