From 3b0394ed6d9135fab7a3851622aef20e1fe4bef1 Mon Sep 17 00:00:00 2001 From: dseditor Date: Wed, 24 Dec 2025 19:50:20 +0800 Subject: [PATCH] fix llama-cpp-auto install --- gguf_inference.py | 221 ++++++++++++++++++++++++++++++---------------- 1 file changed, 143 insertions(+), 78 deletions(-) diff --git a/gguf_inference.py b/gguf_inference.py index 059a2ec..486d811 100644 --- a/gguf_inference.py +++ b/gguf_inference.py @@ -41,6 +41,7 @@ class GGUFInference: self.current_mmproj_path = None self.clip_model_array = None self.llama_cpp_available = False + self.failed_versions = [] # Track failed installation versions self._check_llama_cpp() def _check_llama_cpp(self): @@ -392,11 +393,19 @@ class GGUFInference: print(f"Failed to fetch GitHub releases: {e}") return None - def _find_matching_release(self, releases_list): - """Find matching release and wheel URL from releases list based on system configuration""" + def _find_matching_release(self, releases_list, skip_versions=None): + """Find matching release and wheel URL from releases list based on system configuration + + Args: + releases_list: List of GitHub releases + skip_versions: Set of version tags to skip (previously failed versions) + """ if not releases_list: return None + if skip_versions is None: + skip_versions = set() + # Get system info system = platform.system().lower() # 'windows', 'linux', 'darwin' py_version = sys.version_info @@ -410,6 +419,8 @@ class GGUFInference: print(f" OS: {system}") print(f" Python: {py_version.major}.{py_version.minor} ({py_ver})") print(f" CUDA: {cuda_version if cuda_version else 'Not detected'}") + if skip_versions: + print(f" Skipping versions: {', '.join(sorted(skip_versions))}") print("=" * 70) # Determine platform suffix @@ -430,74 +441,40 @@ class GGUFInference: return None # Build priority list for release matching - # Priority: CUDA Basic > CUDA+AVX2 > CPU Basic > CPU+AVX2 - # Note: Basic versions are prioritized for better compatibility - # AVX2 requires CPU support and may cause "Illegal Instruction" errors + # Priority: CUDA Basic > CPU Basic + # Note: Only use Basic versions (no AVX/AVX2) for maximum compatibility + # AVX and AVX2 versions are skipped due to compatibility issues print("=" * 70) print("Searching through releases...") + print("Note: Skipping AVX/AVX2 versions for better compatibility") print("=" * 70) # Try each priority level priority_filters = [] if cuda_version and system != 'darwin': - # Priority 1: CUDA exact version match (Basic - better compatibility) + # Priority 1: CUDA exact version match (Basic only - no AVX/AVX2) priority_filters.append({ 'name': f'CUDA {cuda_version} Basic (exact match)', 'cuda': cuda_version, - 'avx2': False, - 'exclude_avx2': True, # Only match releases without AVX2 + 'require_basic': True, # Only match Basic (no AVX, no AVX2) 'allow_cuda_upgrade': False }) - # Priority 2: CUDA compatible version (Basic - allows newer CUDA) + # Priority 2: CUDA compatible version (Basic only - allows newer CUDA) # CUDA is forward compatible, e.g., cu130 wheels work on cu128 systems priority_filters.append({ 'name': f'CUDA {cuda_version}+ Basic (compatible)', 'cuda': cuda_version, - 'avx2': False, - 'exclude_avx2': True, - 'allow_cuda_upgrade': True - }) - # Priority 3: CUDA exact version match + AVX2 (requires AVX2 CPU support) - priority_filters.append({ - 'name': f'CUDA {cuda_version} + AVX2 (exact match)', - 'cuda': cuda_version, - 'avx2': True, - 'exclude_avx2': False, - 'allow_cuda_upgrade': False - }) - # Priority 4: CUDA compatible version + AVX2 - priority_filters.append({ - 'name': f'CUDA {cuda_version}+ AVX2 (compatible)', - 'cuda': cuda_version, - 'avx2': True, - 'exclude_avx2': False, + 'require_basic': True, 'allow_cuda_upgrade': True }) - if system != 'darwin': - # Priority 3: CPU Basic version - priority_filters.append({ - 'name': 'CPU Basic', - 'cuda': None, - 'avx2': False, - 'exclude_avx2': True - }) - # Priority 4: CPU AVX2 version - priority_filters.append({ - 'name': 'CPU + AVX2', - 'cuda': None, - 'avx2': True, - 'exclude_avx2': False - }) - else: - # macOS: no AVX2 priority needed - priority_filters.append({ - 'name': 'CPU Basic', - 'cuda': None, - 'avx2': False, - 'exclude_avx2': False - }) + # Priority 3: CPU Basic version (no AVX, no AVX2) + priority_filters.append({ + 'name': 'CPU Basic', + 'cuda': None, + 'require_basic': True + }) # Search through releases with priority for priority in priority_filters: @@ -506,6 +483,10 @@ class GGUFInference: for release in releases_list: release_info = self._parse_release_info(release) + # Skip this version if it's in the skip list + if release_info['tag_name'] in skip_versions: + continue + # Check if this release matches current priority if priority.get('allow_cuda_upgrade', False): # Allow CUDA version upgrade (e.g., cu130 wheel on cu128 system) @@ -525,15 +506,15 @@ class GGUFInference: cuda_match = (priority['cuda'] is None and release_info['cuda_version'] is None) or \ (priority['cuda'] == release_info['cuda_version']) - # Handle exclude_avx2 flag - if priority.get('exclude_avx2', False): - # Only match if release does NOT have AVX2 - avx2_match = not release_info['has_avx2'] + # Handle require_basic flag: skip all AVX and AVX2 versions + if priority.get('require_basic', False): + # Only match Basic versions (no AVX, no AVX2) + basic_match = not release_info['has_avx'] and not release_info['has_avx2'] else: - # Match if either AVX2 is not required or release has AVX2 - avx2_match = (not priority['avx2']) or release_info['has_avx2'] + # Old logic for backward compatibility (not used in new priority system) + basic_match = True - if cuda_match and avx2_match: + if cuda_match and basic_match: # Found matching release, now find wheel for this Python version print(f" Found matching release: {release_info['tag_name']}") @@ -553,7 +534,7 @@ class GGUFInference: print(f" Wheel: {name}") print(f" Download URL: {download_url}") print("=" * 70) - return download_url + return (download_url, release_info['tag_name']) # Release matched but no wheel for this Python version print(f" No wheel found for Python {py_ver} in this release") @@ -564,7 +545,7 @@ class GGUFInference: print("Available releases:") for release in releases_list[:5]: # Show first 5 releases release_info = self._parse_release_info(release) - print(f" - {release_info['tag_name']} (CUDA: {release_info['cuda_version']}, AVX2: {release_info['has_avx2']})") + print(f" - {release_info['tag_name']} (CUDA: {release_info['cuda_version']}, AVX: {release_info['has_avx']}, AVX2: {release_info['has_avx2']})") print("=" * 70) return None @@ -601,12 +582,21 @@ class GGUFInference: print("=" * 70) return False - def _install_llama_cpp(self): + def _install_llama_cpp(self, skip_versions=None): """Auto-installation: Auto-detect system and download from GitHub releases Supports all platforms (Windows/Linux/macOS) and CUDA versions. - Prioritizes Basic (non-AVX2) versions for better compatibility. + Only uses Basic versions (no AVX/AVX2) for maximum compatibility. + + Args: + skip_versions: Set of version tags to skip (previously failed versions) + + Returns: + Tuple of (success: bool, installed_version: str or None) """ + if skip_versions is None: + skip_versions = set() + print("=" * 70) print("Auto-Installing llama-cpp-python") print("Detecting system configuration and searching through releases...") @@ -619,22 +609,25 @@ class GGUFInference: print("ERROR: Failed to fetch releases from GitHub") print("Please visit: https://github.com/JamePeng/llama-cpp-python/releases") print("=" * 70) - return False + return (False, None) # Find matching release and wheel - wheel_url = self._find_matching_release(releases_list) - if not wheel_url: + result = self._find_matching_release(releases_list, skip_versions=skip_versions) + if not result: print("=" * 70) print("ERROR: Could not find a matching wheel for your system") print("Please visit: https://github.com/JamePeng/llama-cpp-python/releases") print("And manually download the appropriate wheel for your system") print("=" * 70) - return False + return (False, None) + + wheel_url, version_tag = result # Download and install try: print("=" * 70) print("Installing llama-cpp-python...") + print(f"Version: {version_tag}") print("=" * 70) subprocess.check_call([ @@ -645,7 +638,7 @@ class GGUFInference: print("SUCCESS: llama-cpp-python installed successfully") print("IMPORTANT: Please restart ComfyUI to use the GGUF node") print("=" * 70) - return True + return (True, version_tag) except Exception as e: print("=" * 70) @@ -653,7 +646,7 @@ class GGUFInference: print("") print("Please visit: https://github.com/JamePeng/llama-cpp-python/releases") print("=" * 70) - return False + return (False, version_tag) def _is_vision_model(self, model_path: str) -> bool: @@ -799,7 +792,8 @@ class GGUFInference: if not self.llama_cpp_available: if auto_install_llama_cpp: print("llama-cpp-python not found, attempting auto-installation...") - if self._install_llama_cpp(): + success, version = self._install_llama_cpp(skip_versions=set(self.failed_versions)) + if success: error_msg = "llama-cpp-python installed successfully!\n\nPlease restart ComfyUI to use the GGUF node." else: error_msg = "Failed to install llama-cpp-python.\n\nPlease visit: https://github.com/JamePeng/llama-cpp-python/releases" @@ -942,30 +936,101 @@ class GGUFInference: enable_vision = False # Load model - if not self._load_model(model_path, enable_vision, mmproj_path): - # Model loading failed + load_result = self._load_model(model_path, enable_vision, mmproj_path) + if not load_result: + # Model loading failed - get the error details from the exception if auto_install_llama_cpp: + # Check for specific error types + import traceback + error_trace = traceback.format_exc() + + # Check for WinError + has_winerror = 'WinError' in error_trace or 'WindowsError' in error_trace + + # Check for ggml.dll error + has_ggml_dll_error = 'ggml.dll' in error_trace.lower() or 'cannot load library' in error_trace.lower() + print("=" * 70) print("ERROR: Model loading failed!") - print("This may indicate llama-cpp-python is incompatible or corrupted.") - print("Attempting to uninstall and reinstall llama-cpp-python...") + + if has_ggml_dll_error: + # Special handling for ggml.dll error + error_msg = ( + "ERROR: Cannot load ggml.dll\n\n" + "This error typically occurs when:\n" + "1. Your CUDA version is incompatible with the installed llama-cpp-python\n" + "2. CUDA runtime libraries are missing or outdated\n\n" + "Recommended solutions:\n" + "1. Update your NVIDIA GPU drivers and CUDA toolkit\n" + "2. Or manually install a compatible llama-cpp-python version from:\n" + " https://github.com/JamePeng/llama-cpp-python/releases\n\n" + "The auto-installer will now try to find an older compatible version..." + ) + print(error_msg) + print("=" * 70) + + # Get current installed version info + try: + import pkg_resources + current_version = pkg_resources.get_distribution("llama-cpp-python").version + print(f"Current llama-cpp-python version: {current_version}") + except: + current_version = None + + if has_winerror or has_ggml_dll_error: + print("Detected compatibility issue - will try previous version") + print("Attempting to uninstall and install previous version...") + else: + print("This may indicate llama-cpp-python is incompatible or corrupted.") + print("Attempting to uninstall and reinstall llama-cpp-python...") print("=" * 70) - # Try to uninstall and reinstall + # Try to uninstall and reinstall with previous version if self._uninstall_llama_cpp(): print("\nNow attempting to install compatible version...") - if self._install_llama_cpp(): - error_msg = "llama-cpp-python has been reinstalled!\n\nIMPORTANT: Please restart ComfyUI to use the new version." + + # Try to install, skipping failed versions + success, new_version = self._install_llama_cpp(skip_versions=set(self.failed_versions)) + + if success and new_version: + # Track this version if it was just installed + if new_version not in self.failed_versions: + self.failed_versions.append(new_version) + + error_msg = f"llama-cpp-python has been reinstalled (version: {new_version})!\n\nIMPORTANT: Please restart ComfyUI to use the new version." print("=" * 70) print(error_msg) print("=" * 70) return (error_msg, seed) else: - error_msg = "Failed to reinstall llama-cpp-python.\n\nPlease visit: https://github.com/JamePeng/llama-cpp-python/releases\nAnd manually install the appropriate version for your system." + # Installation failed - add to failed versions if we got a version + if new_version and new_version not in self.failed_versions: + self.failed_versions.append(new_version) + + error_msg = ( + "Failed to install compatible llama-cpp-python version.\n\n" + "Please visit: https://github.com/JamePeng/llama-cpp-python/releases\n" + "And manually install the appropriate version for your system.\n\n" + ) + if has_ggml_dll_error: + error_msg += ( + "Note: For ggml.dll errors, ensure your CUDA toolkit is up to date:\n" + "- Update NVIDIA GPU drivers\n" + "- Install latest CUDA toolkit from: https://developer.nvidia.com/cuda-downloads" + ) else: - error_msg = "Failed to uninstall llama-cpp-python.\n\nPlease manually uninstall using: pip uninstall llama-cpp-python\nThen enable auto_install_llama_cpp option to reinstall." + error_msg = ( + "Failed to uninstall llama-cpp-python.\n\n" + "Please manually uninstall using: pip uninstall llama-cpp-python\n" + "Then enable auto_install_llama_cpp option to reinstall." + ) else: - error_msg = "Error: Model loading failed.\n\nThis may indicate llama-cpp-python is incompatible.\nPlease enable 'auto_install_llama_cpp' to automatically fix this,\nor manually reinstall llama-cpp-python." + error_msg = ( + "Error: Model loading failed.\n\n" + "This may indicate llama-cpp-python is incompatible.\n" + "Please enable 'auto_install_llama_cpp' to automatically fix this,\n" + "or manually reinstall llama-cpp-python." + ) print(error_msg) return (error_msg, seed)