fix llama-cpp-auto install

This commit is contained in:
dseditor
2025-12-24 19:50:20 +08:00
parent 0c6cf193e1
commit 3b0394ed6d
+143 -78
View File
@@ -41,6 +41,7 @@ class GGUFInference:
self.current_mmproj_path = None
self.clip_model_array = None
self.llama_cpp_available = False
self.failed_versions = [] # Track failed installation versions
self._check_llama_cpp()
def _check_llama_cpp(self):
@@ -392,11 +393,19 @@ class GGUFInference:
print(f"Failed to fetch GitHub releases: {e}")
return None
def _find_matching_release(self, releases_list):
"""Find matching release and wheel URL from releases list based on system configuration"""
def _find_matching_release(self, releases_list, skip_versions=None):
"""Find matching release and wheel URL from releases list based on system configuration
Args:
releases_list: List of GitHub releases
skip_versions: Set of version tags to skip (previously failed versions)
"""
if not releases_list:
return None
if skip_versions is None:
skip_versions = set()
# Get system info
system = platform.system().lower() # 'windows', 'linux', 'darwin'
py_version = sys.version_info
@@ -410,6 +419,8 @@ class GGUFInference:
print(f" OS: {system}")
print(f" Python: {py_version.major}.{py_version.minor} ({py_ver})")
print(f" CUDA: {cuda_version if cuda_version else 'Not detected'}")
if skip_versions:
print(f" Skipping versions: {', '.join(sorted(skip_versions))}")
print("=" * 70)
# Determine platform suffix
@@ -430,74 +441,40 @@ class GGUFInference:
return None
# Build priority list for release matching
# Priority: CUDA Basic > CUDA+AVX2 > CPU Basic > CPU+AVX2
# Note: Basic versions are prioritized for better compatibility
# AVX2 requires CPU support and may cause "Illegal Instruction" errors
# Priority: CUDA Basic > CPU Basic
# Note: Only use Basic versions (no AVX/AVX2) for maximum compatibility
# AVX and AVX2 versions are skipped due to compatibility issues
print("=" * 70)
print("Searching through releases...")
print("Note: Skipping AVX/AVX2 versions for better compatibility")
print("=" * 70)
# Try each priority level
priority_filters = []
if cuda_version and system != 'darwin':
# Priority 1: CUDA exact version match (Basic - better compatibility)
# Priority 1: CUDA exact version match (Basic only - no AVX/AVX2)
priority_filters.append({
'name': f'CUDA {cuda_version} Basic (exact match)',
'cuda': cuda_version,
'avx2': False,
'exclude_avx2': True, # Only match releases without AVX2
'require_basic': True, # Only match Basic (no AVX, no AVX2)
'allow_cuda_upgrade': False
})
# Priority 2: CUDA compatible version (Basic - allows newer CUDA)
# Priority 2: CUDA compatible version (Basic only - allows newer CUDA)
# CUDA is forward compatible, e.g., cu130 wheels work on cu128 systems
priority_filters.append({
'name': f'CUDA {cuda_version}+ Basic (compatible)',
'cuda': cuda_version,
'avx2': False,
'exclude_avx2': True,
'allow_cuda_upgrade': True
})
# Priority 3: CUDA exact version match + AVX2 (requires AVX2 CPU support)
priority_filters.append({
'name': f'CUDA {cuda_version} + AVX2 (exact match)',
'cuda': cuda_version,
'avx2': True,
'exclude_avx2': False,
'allow_cuda_upgrade': False
})
# Priority 4: CUDA compatible version + AVX2
priority_filters.append({
'name': f'CUDA {cuda_version}+ AVX2 (compatible)',
'cuda': cuda_version,
'avx2': True,
'exclude_avx2': False,
'require_basic': True,
'allow_cuda_upgrade': True
})
if system != 'darwin':
# Priority 3: CPU Basic version
priority_filters.append({
'name': 'CPU Basic',
'cuda': None,
'avx2': False,
'exclude_avx2': True
})
# Priority 4: CPU AVX2 version
priority_filters.append({
'name': 'CPU + AVX2',
'cuda': None,
'avx2': True,
'exclude_avx2': False
})
else:
# macOS: no AVX2 priority needed
priority_filters.append({
'name': 'CPU Basic',
'cuda': None,
'avx2': False,
'exclude_avx2': False
})
# Priority 3: CPU Basic version (no AVX, no AVX2)
priority_filters.append({
'name': 'CPU Basic',
'cuda': None,
'require_basic': True
})
# Search through releases with priority
for priority in priority_filters:
@@ -506,6 +483,10 @@ class GGUFInference:
for release in releases_list:
release_info = self._parse_release_info(release)
# Skip this version if it's in the skip list
if release_info['tag_name'] in skip_versions:
continue
# Check if this release matches current priority
if priority.get('allow_cuda_upgrade', False):
# Allow CUDA version upgrade (e.g., cu130 wheel on cu128 system)
@@ -525,15 +506,15 @@ class GGUFInference:
cuda_match = (priority['cuda'] is None and release_info['cuda_version'] is None) or \
(priority['cuda'] == release_info['cuda_version'])
# Handle exclude_avx2 flag
if priority.get('exclude_avx2', False):
# Only match if release does NOT have AVX2
avx2_match = not release_info['has_avx2']
# Handle require_basic flag: skip all AVX and AVX2 versions
if priority.get('require_basic', False):
# Only match Basic versions (no AVX, no AVX2)
basic_match = not release_info['has_avx'] and not release_info['has_avx2']
else:
# Match if either AVX2 is not required or release has AVX2
avx2_match = (not priority['avx2']) or release_info['has_avx2']
# Old logic for backward compatibility (not used in new priority system)
basic_match = True
if cuda_match and avx2_match:
if cuda_match and basic_match:
# Found matching release, now find wheel for this Python version
print(f" Found matching release: {release_info['tag_name']}")
@@ -553,7 +534,7 @@ class GGUFInference:
print(f" Wheel: {name}")
print(f" Download URL: {download_url}")
print("=" * 70)
return download_url
return (download_url, release_info['tag_name'])
# Release matched but no wheel for this Python version
print(f" No wheel found for Python {py_ver} in this release")
@@ -564,7 +545,7 @@ class GGUFInference:
print("Available releases:")
for release in releases_list[:5]: # Show first 5 releases
release_info = self._parse_release_info(release)
print(f" - {release_info['tag_name']} (CUDA: {release_info['cuda_version']}, AVX2: {release_info['has_avx2']})")
print(f" - {release_info['tag_name']} (CUDA: {release_info['cuda_version']}, AVX: {release_info['has_avx']}, AVX2: {release_info['has_avx2']})")
print("=" * 70)
return None
@@ -601,12 +582,21 @@ class GGUFInference:
print("=" * 70)
return False
def _install_llama_cpp(self):
def _install_llama_cpp(self, skip_versions=None):
"""Auto-installation: Auto-detect system and download from GitHub releases
Supports all platforms (Windows/Linux/macOS) and CUDA versions.
Prioritizes Basic (non-AVX2) versions for better compatibility.
Only uses Basic versions (no AVX/AVX2) for maximum compatibility.
Args:
skip_versions: Set of version tags to skip (previously failed versions)
Returns:
Tuple of (success: bool, installed_version: str or None)
"""
if skip_versions is None:
skip_versions = set()
print("=" * 70)
print("Auto-Installing llama-cpp-python")
print("Detecting system configuration and searching through releases...")
@@ -619,22 +609,25 @@ class GGUFInference:
print("ERROR: Failed to fetch releases from GitHub")
print("Please visit: https://github.com/JamePeng/llama-cpp-python/releases")
print("=" * 70)
return False
return (False, None)
# Find matching release and wheel
wheel_url = self._find_matching_release(releases_list)
if not wheel_url:
result = self._find_matching_release(releases_list, skip_versions=skip_versions)
if not result:
print("=" * 70)
print("ERROR: Could not find a matching wheel for your system")
print("Please visit: https://github.com/JamePeng/llama-cpp-python/releases")
print("And manually download the appropriate wheel for your system")
print("=" * 70)
return False
return (False, None)
wheel_url, version_tag = result
# Download and install
try:
print("=" * 70)
print("Installing llama-cpp-python...")
print(f"Version: {version_tag}")
print("=" * 70)
subprocess.check_call([
@@ -645,7 +638,7 @@ class GGUFInference:
print("SUCCESS: llama-cpp-python installed successfully")
print("IMPORTANT: Please restart ComfyUI to use the GGUF node")
print("=" * 70)
return True
return (True, version_tag)
except Exception as e:
print("=" * 70)
@@ -653,7 +646,7 @@ class GGUFInference:
print("")
print("Please visit: https://github.com/JamePeng/llama-cpp-python/releases")
print("=" * 70)
return False
return (False, version_tag)
def _is_vision_model(self, model_path: str) -> bool:
@@ -799,7 +792,8 @@ class GGUFInference:
if not self.llama_cpp_available:
if auto_install_llama_cpp:
print("llama-cpp-python not found, attempting auto-installation...")
if self._install_llama_cpp():
success, version = self._install_llama_cpp(skip_versions=set(self.failed_versions))
if success:
error_msg = "llama-cpp-python installed successfully!\n\nPlease restart ComfyUI to use the GGUF node."
else:
error_msg = "Failed to install llama-cpp-python.\n\nPlease visit: https://github.com/JamePeng/llama-cpp-python/releases"
@@ -942,30 +936,101 @@ class GGUFInference:
enable_vision = False
# Load model
if not self._load_model(model_path, enable_vision, mmproj_path):
# Model loading failed
load_result = self._load_model(model_path, enable_vision, mmproj_path)
if not load_result:
# Model loading failed - get the error details from the exception
if auto_install_llama_cpp:
# Check for specific error types
import traceback
error_trace = traceback.format_exc()
# Check for WinError
has_winerror = 'WinError' in error_trace or 'WindowsError' in error_trace
# Check for ggml.dll error
has_ggml_dll_error = 'ggml.dll' in error_trace.lower() or 'cannot load library' in error_trace.lower()
print("=" * 70)
print("ERROR: Model loading failed!")
print("This may indicate llama-cpp-python is incompatible or corrupted.")
print("Attempting to uninstall and reinstall llama-cpp-python...")
if has_ggml_dll_error:
# Special handling for ggml.dll error
error_msg = (
"ERROR: Cannot load ggml.dll\n\n"
"This error typically occurs when:\n"
"1. Your CUDA version is incompatible with the installed llama-cpp-python\n"
"2. CUDA runtime libraries are missing or outdated\n\n"
"Recommended solutions:\n"
"1. Update your NVIDIA GPU drivers and CUDA toolkit\n"
"2. Or manually install a compatible llama-cpp-python version from:\n"
" https://github.com/JamePeng/llama-cpp-python/releases\n\n"
"The auto-installer will now try to find an older compatible version..."
)
print(error_msg)
print("=" * 70)
# Get current installed version info
try:
import pkg_resources
current_version = pkg_resources.get_distribution("llama-cpp-python").version
print(f"Current llama-cpp-python version: {current_version}")
except:
current_version = None
if has_winerror or has_ggml_dll_error:
print("Detected compatibility issue - will try previous version")
print("Attempting to uninstall and install previous version...")
else:
print("This may indicate llama-cpp-python is incompatible or corrupted.")
print("Attempting to uninstall and reinstall llama-cpp-python...")
print("=" * 70)
# Try to uninstall and reinstall
# Try to uninstall and reinstall with previous version
if self._uninstall_llama_cpp():
print("\nNow attempting to install compatible version...")
if self._install_llama_cpp():
error_msg = "llama-cpp-python has been reinstalled!\n\nIMPORTANT: Please restart ComfyUI to use the new version."
# Try to install, skipping failed versions
success, new_version = self._install_llama_cpp(skip_versions=set(self.failed_versions))
if success and new_version:
# Track this version if it was just installed
if new_version not in self.failed_versions:
self.failed_versions.append(new_version)
error_msg = f"llama-cpp-python has been reinstalled (version: {new_version})!\n\nIMPORTANT: Please restart ComfyUI to use the new version."
print("=" * 70)
print(error_msg)
print("=" * 70)
return (error_msg, seed)
else:
error_msg = "Failed to reinstall llama-cpp-python.\n\nPlease visit: https://github.com/JamePeng/llama-cpp-python/releases\nAnd manually install the appropriate version for your system."
# Installation failed - add to failed versions if we got a version
if new_version and new_version not in self.failed_versions:
self.failed_versions.append(new_version)
error_msg = (
"Failed to install compatible llama-cpp-python version.\n\n"
"Please visit: https://github.com/JamePeng/llama-cpp-python/releases\n"
"And manually install the appropriate version for your system.\n\n"
)
if has_ggml_dll_error:
error_msg += (
"Note: For ggml.dll errors, ensure your CUDA toolkit is up to date:\n"
"- Update NVIDIA GPU drivers\n"
"- Install latest CUDA toolkit from: https://developer.nvidia.com/cuda-downloads"
)
else:
error_msg = "Failed to uninstall llama-cpp-python.\n\nPlease manually uninstall using: pip uninstall llama-cpp-python\nThen enable auto_install_llama_cpp option to reinstall."
error_msg = (
"Failed to uninstall llama-cpp-python.\n\n"
"Please manually uninstall using: pip uninstall llama-cpp-python\n"
"Then enable auto_install_llama_cpp option to reinstall."
)
else:
error_msg = "Error: Model loading failed.\n\nThis may indicate llama-cpp-python is incompatible.\nPlease enable 'auto_install_llama_cpp' to automatically fix this,\nor manually reinstall llama-cpp-python."
error_msg = (
"Error: Model loading failed.\n\n"
"This may indicate llama-cpp-python is incompatible.\n"
"Please enable 'auto_install_llama_cpp' to automatically fix this,\n"
"or manually reinstall llama-cpp-python."
)
print(error_msg)
return (error_msg, seed)