Files
2026-06-19 18:45:55 +08:00

403 lines
16 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
GPU显存清理脚本(强化版)
在任务完成后调用,彻底释放GPU显存和内存
支持强制杀死占用GPU的Python进程
"""
import sys
import subprocess
import os
import gc
import time
def get_gpu_processes():
"""获取占用GPU的进程列表"""
try:
result = subprocess.run(
"nvidia-smi --query-compute-apps=pid,process_name,used_memory --format=csv,noheader",
shell=True,
capture_output=True,
text=True,
timeout=5
)
processes = []
if result.returncode == 0 and result.stdout.strip():
for line in result.stdout.strip().split('\n'):
if line.strip():
parts = [p.strip() for p in line.split(',')]
if len(parts) >= 3:
pid, name, memory = parts
processes.append({
'pid': int(pid),
'name': name,
'memory': memory
})
return processes
except Exception as e:
print("[GPU清理] 获取GPU进程失表: {}".format(e))
return []
def get_all_python_processes():
"""获取所有Python进程列表(包括孤儿进程)"""
try:
result = subprocess.run(
"tasklist /FI \"IMAGENAME eq python.exe\" /FO CSV /NH",
shell=True,
capture_output=True,
text=True,
timeout=5
)
processes = []
if result.returncode == 0 and result.stdout.strip():
for line in result.stdout.strip().split('\n'):
if line.strip():
parts = [p.strip('"') for p in line.split(',')]
if len(parts) >= 2:
name = parts[0]
pid = parts[1]
try:
processes.append({
'pid': int(pid),
'name': name
})
except:
pass
return processes
except Exception as e:
print("[GPU清理] 获取Python进程失败: {}".format(e))
return []
def should_kill_process(pid, model_paths=None):
"""判断进程是否应该被清理(只清理AI模型相关进程)
Args:
pid: 进程ID
model_paths: 模型路径列表,如果提供,只清理这些路径下的进程
"""
try:
# 获取进程命令行
cmdline_result = subprocess.run(
"wmic process where ProcessId={} get CommandLine /format:list".format(pid),
shell=True,
capture_output=True,
text=True,
timeout=2
)
if cmdline_result.returncode != 0 or not cmdline_result.stdout:
return False
cmdline = cmdline_result.stdout.lower()
# 白名单(包含这些关键词的进程不清理)
whitelist_keywords = [
'直播伴侣',
'finderliveobs',
'xwechat',
'obs',
'obs-studio',
'streamlabs',
'douyin',
'tiktok',
'bilibili',
'抖音',
'kuaishou',
'快手',
'pycharm',
'vscode',
'visual studio code',
'jupyter',
'spyder',
'anaconda',
'conda',
'sublime',
'notepad++',
'wechat',
'weixin',
'微信',
'qq',
'asr',
'funasr',
'aigcpanel',
'whisper',
'cosyvoice',
]
# 检查白名单
for keyword in whitelist_keywords:
if keyword in cmdline:
# 提取进程名称(从命令行中获取)
process_name = "未知"
if "python" in cmdline:
process_name = "Python进程"
elif "finderliveobs" in cmdline or "xwechat" in cmdline:
process_name = "视频号直播伴侣"
elif "obs" in cmdline:
process_name = "OBS直播软件"
elif "wechat" in cmdline or "weixin" in cmdline:
process_name = "微信"
print("[GPU清理] [保护] 跳过白名单进程 PID={} ({}, 匹配关键词: {})".format(pid, process_name, keyword))
return False
# 🔧 必须提供模型路径才进行清理(安全模式)
if model_paths and len(model_paths) > 0:
for model_path in model_paths:
# 标准化路径格式(统一使用小写和反斜杠)
normalized_path = model_path.lower().replace('/', '\\')
if normalized_path in cmdline:
print("[GPU清理] 匹配模型路径 PID={} (路径: {})".format(pid, model_path))
return True
# 如果提供了路径但都不匹配,不清理
return False
else:
# 🔧 安全修复:没有提供路径时,不清理任何进程(防止误杀)
# 之前的关键词匹配模式已移除,确保只清理模型路径下的进程
print("[GPU清理] 警告: 未提供模型路径,跳过进程 PID={}".format(pid))
return False
except Exception as e:
print("[GPU清理] 检查进程 PID={} 失败: {}".format(pid, e))
return False
def cleanup_gpu(force_kill=True, model_paths=None):
"""彻底清理GPU显存和内存(默认强力模式)
Args:
force_kill: 是否强制清理
model_paths: 模型路径列表,只清理这些路径下的进程
"""
try:
print("[GPU清理] ========== 开始GPU显存和内存清理 ==========")
if model_paths and len(model_paths) > 0:
print("[GPU清理] [模式] 路径过滤:只清理模型路径下的Python进程")
print("[GPU清理] [路径] 共 {} 个模型路径:".format(len(model_paths)))
for path in model_paths:
print("[GPU清理] - {}".format(path))
else:
print("[GPU清理] [模式] 智能清理:只清理AI模型进程,保护其他Python应用")
# ==================== 步骤1PyTorch GPU深度清理 ====================
print("[GPU清理] [步骤1] PyTorch GPU深度清理...")
try:
import torch
if torch.cuda.is_available():
# 显示清理前的显存状态
allocated_before = torch.cuda.memory_allocated() / 1024**3
reserved_before = torch.cuda.memory_reserved() / 1024**3
print("[GPU清理] 清理前: 已分配={:.2f}GB, 已保留={:.2f}GB".format(allocated_before, reserved_before))
# 1. 清空CUDA缓存(多次循环确保彻底清理)
print("[GPU清理] 正在清空CUDA缓存...")
for i in range(5):
torch.cuda.empty_cache()
gc.collect()
time.sleep(0.1)
# 2. 同步所有CUDA设备
print("[GPU清理] 同步CUDA设备...")
torch.cuda.synchronize()
# 3. 重置内存统计
torch.cuda.reset_peak_memory_stats()
torch.cuda.reset_accumulated_memory_stats()
# 4. 尝试释放所有未使用的缓存内存
try:
# 清理内存分配器的缓存
torch.cuda.memory.empty_cache()
except:
pass
# 5. 设置内存分配器配置(优化内存使用)
try:
os.environ['PYTORCH_CUDA_ALLOC_CONF'] = 'max_split_size_mb:512'
except:
pass
# 显示清理后的显存状态
allocated_after = torch.cuda.memory_allocated() / 1024**3
reserved_after = torch.cuda.memory_reserved() / 1024**3
freed_memory = (reserved_before - reserved_after)
print("[GPU清理] 清理后: 已分配={:.2f}GB, 已保留={:.2f}GB".format(allocated_after, reserved_after))
print("[GPU清理] 释放显存: {:.2f}GB".format(freed_memory))
print("[GPU清理] [OK] PyTorch GPU清理完成")
else:
print("[GPU清理] CUDA不可用,跳过GPU清理")
except ImportError:
print("[GPU清理] PyTorch未安装,跳过GPU清理")
except Exception as e:
print("[GPU清理] PyTorch清理出错: {}".format(e))
# ==================== 步骤2:强制系统内存清理 ====================
print("[GPU清理] [步骤2] 执行强制系统内存清理...")
# 1. 多轮垃圾回收(清理不同代的对象)
total_collected = 0
for i in range(5):
collected = gc.collect(generation=2) # 清理所有代
total_collected += collected
time.sleep(0.05)
print("[GPU清理] 垃圾回收: 清理了 {} 个对象".format(total_collected))
# 2. 清理Python内部缓存
try:
import ctypes
# 尝试释放Python未使用的内存回操作系统
if hasattr(ctypes, 'windll'):
# Windows平台
ctypes.windll.kernel32.SetProcessWorkingSetSize(-1, -1, -1)
except:
pass
print("[GPU清理] [OK] 系统内存清理完成")
# ==================== 步骤3:显示当前状态 ====================
print("[GPU清理] [步骤3] 检查GPU状态...")
try:
result = subprocess.run(
"nvidia-smi --query-gpu=index,memory.used,memory.free,memory.total --format=csv,noheader",
shell=True,
capture_output=True,
text=True,
timeout=5
)
if result.returncode == 0:
print("[GPU清理] ========== GPU 显存状态 ==========")
for line in result.stdout.strip().split('\n'):
if line.strip():
print("[GPU清理] {}".format(line.strip()))
except:
pass
# ==================== 步骤4:清理所有Python进程(包括孤儿进程) ====================
print("[GPU清理] [步骤4] 检查并清理所有Python进程...")
# 获取所有Python进程(包括GPU和非GPU进程)
all_python_processes = get_all_python_processes()
if all_python_processes:
print("[GPU清理] 检测到 {} 个Python进程(包括AI任务和孤儿进程)".format(len(all_python_processes)))
# 过滤需要清理的进程
print("[GPU清理] [智能过滤] 正在识别需要清理的进程...")
processes_to_kill = []
skipped_count = 0
for proc in all_python_processes:
if should_kill_process(proc['pid'], model_paths):
processes_to_kill.append(proc)
print("[GPU清理] PID={}, 将被清理 (AI模型进程)".format(proc['pid']))
else:
skipped_count += 1
print("[GPU清理] 识别完成: {} 个进程将被清理, {} 个进程已跳过".format(
len(processes_to_kill), skipped_count
))
# 清理识别出的AI模型进程
if processes_to_kill:
print("[GPU清理] [智能清理] 开始清理AI模型进程...")
killed_count = 0
failed_pids = []
for proc in processes_to_kill:
try:
# 使用 PowerShell 强制杀死进程(更可靠)
result = subprocess.run(
'powershell -Command "Stop-Process -Id {} -Force"'.format(proc['pid']),
shell=True,
capture_output=True,
timeout=3
)
if result.returncode == 0:
print("[GPU清理] [OK] 已清理进程 PID={}".format(proc['pid']))
killed_count += 1
else:
failed_pids.append(proc['pid'])
print("[GPU清理] [FAIL] 清理进程 PID={} 失败".format(proc['pid']))
time.sleep(0.2)
except Exception as e:
failed_pids.append(proc['pid'])
print("[GPU清理] [ERROR] 清理进程PID={} 异常: {}".format(proc['pid'], str(e)))
print("[GPU清理] [完成] 成功清理 {} 个进程, 失败 {}".format(killed_count, len(failed_pids)))
else:
print("[GPU清理] [OK] 没有需要清理的AI模型进程")
# 清理后再次检查
time.sleep(1)
remaining = get_all_python_processes()
remaining_ai_processes = [p for p in remaining if should_kill_process(p['pid'], model_paths)]
if not remaining_ai_processes:
print("[GPU清理] [OK] AI模型进程已清理完成")
else:
print("[GPU清理] [提示] 仍有 {} 个AI模型进程存在".format(len(remaining_ai_processes)))
for proc in remaining_ai_processes:
print("[GPU清理] 残留进程 PID={}".format(proc['pid']))
else:
print("[GPU清理] [OK] 没有检测到Python进程")
# ==================== 步骤5:最终GPU显存清理 ====================
print("[GPU清理] [步骤5] 最终GPU显存清理...")
try:
import torch
if torch.cuda.is_available():
# 进程清理后,再次清理GPU缓存
for i in range(3):
torch.cuda.empty_cache()
gc.collect()
time.sleep(0.1)
# 显示最终状态
final_allocated = torch.cuda.memory_allocated() / 1024**3
final_reserved = torch.cuda.memory_reserved() / 1024**3
print("[GPU清理] 最终状态: 已分配={:.2f}GB, 已保留={:.2f}GB".format(final_allocated, final_reserved))
print("[GPU清理] [OK] 最终清理完成")
else:
print("[GPU清理] CUDA不可用")
except:
pass
print("[GPU清理] ========== GPU和内存清理完成 ==========\n")
return True
except Exception as e:
print("[GPU清理] [ERROR] 清理失败: {}".format(str(e)))
import traceback
traceback.print_exc()
return False
if __name__ == "__main__":
# 解析命令行参数获取模型路径
model_paths = []
if len(sys.argv) > 1:
# 从命令行参数获取模型路径(以逗号分隔)
paths_arg = sys.argv[1]
if paths_arg and paths_arg.strip():
model_paths = [p.strip() for p in paths_arg.split(',') if p.strip()]
if model_paths:
print("[GPU清理] [信息] 路径过滤模式:只清理指定模型路径下的Python进程")
print("[GPU清理] [信息] 受保护的应用:直播伴侣、OBS、IDE等不会被清理")
print("[GPU清理] [信息] 模型路径数: {}\n".format(len(model_paths)))
else:
print("[GPU清理] [警告] 未提供模型路径参数!")
print("[GPU清理] [说明] 脚本仅在提供模型路径时才进行进程清理")
print("[GPU清理] [说明] 用法: python cleanup_gpu.py \"path1,path2,path3\"")
print("[GPU清理] [信息] 将执行GPU显存和系统内存清理(不清理任何进程)\n")
cleanup_gpu(force_kill=True, model_paths=model_paths)
sys.exit(0)