This commit introduces a major architectural refactoring, laying the groundwork for DisTorch V2. The changes focus on improving modularity, memory management, and diagnostics. Key changes include: - Renaming `distorch_safetensor.py` to `distorch_2.py` to house the new core logic. - Deleting the legacy `block_swap.py` module. - Adding `device_memory_audit.py` for more sophisticated analysis of GPU memory usage. - Implementing a centralized and configurable logging system in `__init__.py` to provide standardized and level-controlled (DEBUG/INFO) output for better debugging.
28 lines
1.1 KiB
Python
28 lines
1.1 KiB
Python
import logging
|
|
# Utility to get memory stats, using self-contained hardware_info module.
|
|
|
|
def log_memory_usage(label=""):
|
|
try:
|
|
from .hardware_info import CHardwareInfo
|
|
|
|
hardware_info = CHardwareInfo(switchRAM=True, switchVRAM=True)
|
|
status = hardware_info.getStatus()
|
|
|
|
ram_used_gb = status.get('ram_used', 0) / (1024**3)
|
|
ram_total_gb = status.get('ram_total', 0) / (1024**3)
|
|
|
|
log_message = f"[MultiGPU] {label} | RAM Used: {ram_used_gb:.2f}/{ram_total_gb:.2f} GB"
|
|
|
|
if 'gpus' in status:
|
|
for i, gpu in enumerate(status['gpus']):
|
|
vram_used_gb = gpu.get('vram_used', 0) / (1024**3)
|
|
vram_total_gb = gpu.get('vram_total', 0) / (1024**3)
|
|
log_message += f" | VRAM cuda:{i}: {vram_used_gb:.2f}/{vram_total_gb:.2f} GB"
|
|
|
|
logging.debug(log_message)
|
|
|
|
except ImportError:
|
|
logging.warning("[MultiGPU] Could not import local CHardwareInfo. Cannot log memory usage.")
|
|
except Exception as e:
|
|
logging.error(f"[MultiGPU] Error getting memory usage: {e}")
|