commit 680e004bf034008e9a44acda8f05c574629d7686 Author: Sinphaltimus Date: Fri May 9 07:42:44 2025 -0400 First Commit diff --git a/LICENSE.txt b/LICENSE.txt new file mode 100644 index 0000000..8b8f87f --- /dev/null +++ b/LICENSE.txt @@ -0,0 +1,9 @@ +MIT License + +Copyright (c) 2025 Reverend Pope of the Poconos Doktor Sinphaltimus Exmortus of the First Ever Digital Church Of Mind Slack, Destroyer of Chairs, Splitter of Aircraft, Raiser of Packs, Pixel Pushing, Sound Dabbling, AI Fondling Enabler of AiRTwerx, Yeti Bellowing, Ambassador of Slack. + +Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the "Software"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS," WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE, AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES, OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT, OR OTHERWISE, ARISING FROM, OUT OF, OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/PSscript/Model_Lister_with_paths.ps1 b/PSscript/Model_Lister_with_paths.ps1 new file mode 100644 index 0000000..54d8a01 --- /dev/null +++ b/PSscript/Model_Lister_with_paths.ps1 @@ -0,0 +1,55 @@ +# Function to validate user-input paths +function Get-ValidPath($prompt) { + do { + $path = Read-Host $prompt + if (Test-Path $path) { + return $path + } else { + Write-Host "❌ Invalid path. Please enter a valid directory." -ForegroundColor Red + } + } while ($true) +} + +# Function to filter directories that contain files +function Get-FullPaths($source) { + Get-ChildItem -Path $source -Recurse | Where-Object { $_.PSIsContainer -eq $false } | ForEach-Object { $_.FullName } +} + +# Get valid model directory path from user +$source = Get-ValidPath "Please enter the location of your models directory (eg. c:\ComfyUI_windows_portable\ComfyUI\models):" + +# Get valid save destination from user +$destination = Get-ValidPath "Please enter the location where you would like to save your model_list.txt file (eg. c:\temp):" + +# Generate output file path +$mlistPath = "$destination\model_list.txt" + +# Check if model_list.txt already exists +if (Test-Path $mlistPath) { + do { + Write-Host "`n⚠️ model_list.txt already exists at $mlistPath" + $choice = Read-Host "Do you want to overwrite it? (Y/N)" + + switch ($choice.ToUpper()) { + "Y" { + Write-Host "✏️ Overwriting existing file..." -ForegroundColor Yellow + Remove-Item $mlistPath -Force + break + } + "N" { + $destination = Get-ValidPath "Please enter a new save location:" + $mlistPath = "$destination\model_list.txt" + break + } + default { + Write-Host "❌ Invalid choice! Please enter Y or N." -ForegroundColor Red + } + } + } while ($choice -notmatch "^[YN]$") +} + +# Generate the model list excluding empty directories +Write-Host "`n📂 Scanning directory: $source" +Get-FullPaths $source | Out-File $mlistPath + +Write-Host "`n✅ Model list saved successfully at: $mlistPath" -ForegroundColor Green diff --git a/PSscript/readme_psscript.txt b/PSscript/readme_psscript.txt new file mode 100644 index 0000000..e9ea230 --- /dev/null +++ b/PSscript/readme_psscript.txt @@ -0,0 +1,2 @@ +This is a simple powershell script to generate a model_list file for you to copy paste from. +When I get better with Python, I plan to have better functionality built into the nodes themselves. \ No newline at end of file diff --git a/README.md b/README.md new file mode 100644 index 0000000..a3d8910 --- /dev/null +++ b/README.md @@ -0,0 +1,57 @@ +Readme.md + +Workflow Guide: Extracting Model Metadata + +This workflow begins with running Model_Lister_with_paths.ps1, which lists all available model files along with their paths. Use this output to copy-paste the file paths into each node above for metadata extraction. + +You can preload requirements if you like. +.\ComfyUI_windows_portable\python_embeded\python.exe -m pip install -r requirements.txt +(For Windows ComfyUI Portable as an example) + +Simply copy from the mocel_list.txt file and paste it into one or all nodes above. Connect the String Outputs from anyone of the three nodes to the string input connector of the Display String Node. + +Click RUN and wait. As you progess to Enahnced and Advanced nodes, the data extraction times will increase. Also, for large models, expect log extraction times. Please be patient and let the workflow finish. + +1️⃣ Model Metadata Reader +Purpose: + +Extract basic metadata from models in various formats, including Safetensors, Checkpoints (.ckpt, .pth, .pt, .bin). + +Provides a structured metadata report for supported model formats. + +Detects the model format automatically and applies the correct extraction method. + +How It Works: ✔ Reads metadata from Safetensors models using safetensors.safe_open(). ✔ Extracts available keys from Torch-based models (ckpt, .bin, .pth). ✔ Returns structured metadata when available, otherwise reports unsupported formats. ✔ Logs errors in case extraction fails. + +Use this node for a quick overview of model metadata without deep metadata parsing. + +2️⃣ Enhanced Model Metadata Reader +Purpose: + +Extract deep metadata from models, including structured attributes and raw text parsing. + +Focuses heavily on ONNX models, using direct binary parsing to retrieve metadata without relying on the ONNX Python package. + +How It Works: ✔ Reads ONNX files as raw binary, searching for readable metadata like author, description, version, etc. ✔ Extracts ASCII-readable strings directly from the binary file if structured metadata isn't available. ✔ Provides warnings when metadata is missing but still displays raw extracted text. ✔ Enhanced logging for debugging failed extractions and unsupported formats. + +This node is ideal for ONNX models, offering both metadata and raw text extraction for deeper insights. + +3️⃣ Advanced Model Data Extractor +Purpose: + +Extract structured metadata and raw text together from various model formats. + +Supports Safetensors, Checkpoints (.ckpt, .pth, .bin, .gguf, .onnx). + +How It Works: ✔ Extracts metadata for Safetensors using direct access to model properties. ✔ Retrieves Torch model metadata such as available keys. ✔ Attempts raw text extraction from the binary file using character encoding detection (chardet). ✔ Limits raw text output for readability while keeping detailed extraction logs. + +This node provides both metadata and raw text from models, making it the most comprehensive extraction tool in the workflow. + +🚀 Final Notes +Run Model_Lister_with_paths.ps1 first, then copy a model path into each node. + +Use ModelMetadataReader for quick metadata lookup. + +Use EnhancedModelMetadataReader for deep metadata parsing, especially for ONNX models. + +Use AdvancedModelDataExtractor for full metadata + raw text extraction. \ No newline at end of file diff --git a/Workflow/FEDCOMS-ModelDataInfo_Example.json b/Workflow/FEDCOMS-ModelDataInfo_Example.json new file mode 100644 index 0000000..a2c25a6 --- /dev/null +++ b/Workflow/FEDCOMS-ModelDataInfo_Example.json @@ -0,0 +1 @@ +{"id":"4b7013ea-0520-4cb9-8c91-da2aaf35bb5a","revision":0,"last_node_id":29,"last_link_id":14,"nodes":[{"id":25,"type":"ModelDataExtractor","pos":[-1195.51171875,389.5708923339844],"size":[365.4000244140625,58],"flags":{},"order":0,"mode":0,"inputs":[{"localized_name":"model_path","name":"model_path","type":"STRING","widget":{"name":"model_path"},"link":null}],"outputs":[{"localized_name":"STRING","name":"STRING","type":"STRING","links":[]}],"properties":{"Node name for S&R":"ModelDataExtractor"},"widgets_values":["D:\\ai\\ComfyUI_windows_portable\\ComfyUI\\models\\inswapper_128.onnx"]},{"id":24,"type":"ModelMetadataReader","pos":[-1188.88818359375,151.13705444335938],"size":[358.05047607421875,58],"flags":{},"order":1,"mode":0,"inputs":[{"localized_name":"model_path","name":"model_path","type":"STRING","widget":{"name":"model_path"},"link":null}],"outputs":[{"localized_name":"STRING","name":"STRING","type":"STRING","links":[]}],"properties":{"Node name for S&R":"ModelMetadataReader"},"widgets_values":["D:\\ai\\ComfyUI_windows_portable\\ComfyUI\\models\\inswapper_128.onnx"]},{"id":7,"type":"LF_DisplayString","pos":[-738.5134887695312,180.94131469726562],"size":[662.815185546875,1074.5087890625],"flags":{},"order":4,"mode":0,"inputs":[{"localized_name":"string","name":"string","type":"STRING","link":14},{"localized_name":"ui_widget","name":"ui_widget","shape":7,"type":"LF_CODE","widget":{"name":"ui_widget"},"link":null}],"outputs":[{"localized_name":"string","name":"string","type":"STRING","links":null}],"title":"Extracted Model Info (Display string).","properties":{"cnr_id":"lf-nodes","ver":"561a9953186548321e6d13ed23fd642d4c734f01","Node name for S&R":"LF_DisplayString"},"widgets_values":["📌 Starting Metadata Extraction: 2025-05-08 19:34:06\n🔎 Checking model path: D:\\ai\\ComfyUI_windows_portable\\ComfyUI\\models\\inswapper_128.onnx\n⚠ Unsupported model format detected.\n✅ Extraction Complete: 2025-05-08 19:34:06\n\n{\n \"error\": \"Unsupported model format\"\n}"]},{"id":27,"type":"Note","pos":[-1341.2215576171875,486.7104797363281],"size":[588.1671752929688,770.885986328125],"flags":{},"order":2,"mode":0,"inputs":[],"outputs":[],"title":"Read Me","properties":{},"widgets_values":["Workflow Guide: Extracting Model Metadata\n\nThis workflow begins with running Model_Lister_with_paths.ps1, which lists all available model files along with their paths. Use this output to copy-paste the file paths into each node above for metadata extraction.\n\nYou can preload requirements if you like.\n.\\ComfyUI_windows_portable\\python_embeded\\python.exe -m pip install -r requirements.txt\n(For Windows ComfyUI Portable as an example)\n\nSimply copy from the mocel_list.txt file and paste it into one or all nodes above. Connect the String Outputs from anyone of the three nodes to the string input connector of the Display String Node.\n\n1️⃣ Model Metadata Reader\nPurpose:\n\nExtract basic metadata from models in various formats, including Safetensors, Checkpoints (.ckpt, .pth, .pt, .bin).\n\nProvides a structured metadata report for supported model formats.\n\nDetects the model format automatically and applies the correct extraction method.\n\nHow It Works: ✔ Reads metadata from Safetensors models using safetensors.safe_open(). ✔ Extracts available keys from Torch-based models (ckpt, .bin, .pth). ✔ Returns structured metadata when available, otherwise reports unsupported formats. ✔ Logs errors in case extraction fails.\n\nUse this node for a quick overview of model metadata without deep metadata parsing.\n\n2️⃣ Enhanced Model Metadata Reader\nPurpose:\n\nExtract deep metadata from models, including structured attributes and raw text parsing.\n\nFocuses heavily on ONNX models, using direct binary parsing to retrieve metadata without relying on the ONNX Python package.\n\nHow It Works: ✔ Reads ONNX files as raw binary, searching for readable metadata like author, description, version, etc. ✔ Extracts ASCII-readable strings directly from the binary file if structured metadata isn't available. ✔ Provides warnings when metadata is missing but still displays raw extracted text. ✔ Enhanced logging for debugging failed extractions and unsupported formats.\n\nThis node is ideal for ONNX models, offering both metadata and raw text extraction for deeper insights.\n\n3️⃣ Advanced Model Data Extractor\nPurpose:\n\nExtract structured metadata and raw text together from various model formats.\n\nSupports Safetensors, Checkpoints (.ckpt, .pth, .bin, .gguf, .onnx).\n\nHow It Works: ✔ Extracts metadata for Safetensors using direct access to model properties. ✔ Retrieves Torch model metadata such as available keys. ✔ Attempts raw text extraction from the binary file using character encoding detection (chardet). ✔ Limits raw text output for readability while keeping detailed extraction logs.\n\nThis node provides both metadata and raw text from models, making it the most comprehensive extraction tool in the workflow.\n\n🚀 Final Notes\nRun Model_Lister_with_paths.ps1 first, then copy a model path into each node.\n\nUse ModelMetadataReader for quick metadata lookup.\n\nUse EnhancedModelMetadataReader for deep metadata parsing, especially for ONNX models.\n\nUse AdvancedModelDataExtractor for full metadata + raw text extraction."],"color":"#432","bgcolor":"#653"},{"id":26,"type":"EnhancedModelMetadataReader","pos":[-1189.9920654296875,268.146240234375],"size":[352.6112365722656,58],"flags":{},"order":3,"mode":0,"inputs":[{"localized_name":"model_path","name":"model_path","type":"STRING","widget":{"name":"model_path"},"link":null}],"outputs":[{"localized_name":"STRING","name":"STRING","type":"STRING","links":[14]}],"properties":{"Node name for S&R":"EnhancedModelMetadataReader"},"widgets_values":["D:\\ai\\ComfyUI_windows_portable\\ComfyUI\\models\\inswapper_128.onnx"]}],"links":[[14,26,0,7,0,"STRING"]],"groups":[],"config":{},"extra":{"ds":{"scale":0.7972024500000047,"offset":[1561.714277530711,-92.26897791416417]},"frontendVersion":"1.17.11"},"version":0.4} \ No newline at end of file diff --git a/__init__.py b/__init__.py new file mode 100644 index 0000000..f4cbf3e --- /dev/null +++ b/__init__.py @@ -0,0 +1,15 @@ +from .modelmetadatareader import ModelMetadataReader +from .modeldataextractor import ModelDataExtractor +from .enhancedmodelmetadatareader import EnhancedModelMetadataReader # ✅ New Node Added + +NODE_CLASS_MAPPINGS = { + "ModelMetadataReader": ModelMetadataReader, + "ModelDataExtractor": ModelDataExtractor, + "EnhancedModelMetadataReader": EnhancedModelMetadataReader, # ✅ Third Node Included +} + +NODE_DISPLAY_NAME_MAPPINGS = { + "ModelMetadataReader": "Model Metadata Reader", + "ModelDataExtractor": "Advanced Model Data Extractor", + "EnhancedModelMetadataReader": "Enhanced Model Metadata Reader", # ✅ Custom UI Label +} diff --git a/enhancedmodelmetadatareader.py b/enhancedmodelmetadatareader.py new file mode 100644 index 0000000..f858db1 --- /dev/null +++ b/enhancedmodelmetadatareader.py @@ -0,0 +1,67 @@ +import os +import re +import json +import datetime + +class EnhancedModelMetadataReader: + CATEGORY = "Model Tools" + + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "model_path": ("STRING", {"default": "Enter full model path here", "trigger": True}), + } + } + + RETURN_TYPES = ("STRING",) + FUNCTION = "extract_data" + + def extract_data(self, model_path): + """Extracts metadata and raw text from an ONNX file with enhanced logging.""" + + log_entries = [] + log_entries.append(f"📌 Starting Metadata Extraction: {self.get_timestamp()}") + log_entries.append(f"🔎 Checking model path: {model_path}") + + if not os.path.exists(model_path): + log_entries.append("❌ Error: Model not found.") + return ("\n".join(log_entries),) + + # ✅ Extract Metadata and Raw Text from ONNX + metadata, raw_text = self.extract_onnx_metadata(model_path) + log_entries.append("📂 Metadata extraction method: Direct Binary Parsing") + + log_entries.append(f"✅ Extraction Complete: {self.get_timestamp()}") + + return ("\n".join(log_entries) + "\n\n" + json.dumps(metadata, indent=4) + "\n\n🔍 Extracted Raw Text:\n" + raw_text[:2000],) + + def extract_onnx_metadata(self, file_path): + """Extracts readable metadata and raw text from an ONNX file using direct binary parsing.""" + try: + with open(file_path, "rb") as f: + data = f.read() + + # Extract human-readable text sections + extracted_text = re.findall(rb'[ -~]{4,}', data) # Captures ASCII-readable characters + decoded_text = [text.decode("utf-8", errors="ignore") for text in extracted_text] + + # Look for potential metadata-related fields + metadata_keys = ["author", "description", "license", "version", "model_name"] + metadata_found = {key: value for value in decoded_text if any(key in value.lower() for key in metadata_keys)} + + # Combine raw extracted text + raw_text_output = "\n".join(decoded_text) + + return metadata_found if metadata_found else {"warning": "No structured metadata found."}, raw_text_output + + except Exception as e: + return {"error": f"Failed to extract metadata: {str(e)}"}, "" + + def get_timestamp(self): + """Returns formatted timestamp.""" + return datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S") + +NODE_CLASS_MAPPINGS = { + "EnhancedModelMetadataReader": EnhancedModelMetadataReader +} diff --git a/modeldataextractor.py b/modeldataextractor.py new file mode 100644 index 0000000..8b702ea --- /dev/null +++ b/modeldataextractor.py @@ -0,0 +1,85 @@ +import os +import json +import torch +import safetensors +import chardet +import datetime + +class ModelDataExtractor: + CATEGORY = "Model Tools" + + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "model_path": ("STRING", {"default": "Enter full model path here", "trigger": True}), + } + } + + RETURN_TYPES = ("STRING",) + FUNCTION = "extract_data" + + def extract_data(self, model_path): + """Extracts metadata and raw text with enhanced logging.""" + + log_entries = [] + log_entries.append(f"📌 Starting Data Extraction: {self.get_timestamp()}") + log_entries.append(f"🔎 Checking model path: {model_path}") + + if not os.path.exists(model_path): + log_entries.append("❌ Error: Model not found.") + return ("\n".join(log_entries),) + + extracted_data = {} + + # ✅ Extract structured metadata based on model format + if model_path.endswith(".safetensors"): + extracted_data["structured_metadata"] = self.read_safetensors_metadata(model_path) + log_entries.append("📂 Metadata extraction method: Safetensors") + elif model_path.endswith((".ckpt", ".pth", ".pt", ".bin", ".gguf", ".onnx")): + extracted_data["structured_metadata"] = self.read_torch_metadata(model_path) + log_entries.append("📂 Metadata extraction method: Checkpoint/Torch") + + # ✅ Always attempt raw text extraction + extracted_data["raw_text"] = self.extract_raw_text(model_path) + log_entries.append("📂 Attempting raw text extraction.") + + log_entries.append(f"✅ Extraction Complete: {self.get_timestamp()}") + + return ("\n".join(log_entries) + "\n\n" + json.dumps(extracted_data, indent=4),) + + def read_safetensors_metadata(self, model_path): + """Reads metadata from Safetensors models.""" + try: + with safetensors.safe_open(model_path, framework="pt") as f: + return f.metadata() + except Exception as e: + return {"error": f"Safetensors extraction failed: {str(e)}"} + + def read_torch_metadata(self, model_path): + """Reads metadata from Torch-based models.""" + try: + model_data = torch.load(model_path, map_location="cpu") + return {"metadata_keys": list(model_data.keys())} + except Exception as e: + return {"error": f"Torch model extraction failed: {str(e)}"} + + def extract_raw_text(self, model_path): + """Attempts raw text extraction from model binaries.""" + try: + with open(model_path, "rb") as f: + data = f.read() + encoding = chardet.detect(data)["encoding"] + text_data = data.decode(encoding, errors="ignore") if encoding else "Encoding not detected" + + return text_data[:2000] # ✅ Increased limit for better visibility + except Exception as e: + return f"Error extracting raw text: {str(e)}" + + def get_timestamp(self): + """Returns formatted timestamp.""" + return datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S") + +NODE_CLASS_MAPPINGS = { + "ModelDataExtractor": ModelDataExtractor +} diff --git a/modelmetadatareader.py b/modelmetadatareader.py new file mode 100644 index 0000000..8b0b6c1 --- /dev/null +++ b/modelmetadatareader.py @@ -0,0 +1,71 @@ +import os +import json +import torch +import safetensors +import datetime + +class ModelMetadataReader: + CATEGORY = "Model Tools" + + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "model_path": ("STRING", {"default": "Enter full model path here", "trigger": True}), + } + } + + RETURN_TYPES = ("STRING",) + FUNCTION = "get_metadata" + + def get_metadata(self, model_path): + """Retrieves metadata from a model file with enhanced logging.""" + + log_entries = [] + log_entries.append(f"📌 Starting Metadata Extraction: {self.get_timestamp()}") + log_entries.append(f"🔎 Checking model path: {model_path}") + + if not os.path.exists(model_path): + log_entries.append("❌ Error: Model not found.") + return ("\n".join(log_entries),) + + metadata = {} + + # ✅ Extract metadata based on file type + if model_path.endswith(".safetensors"): + metadata = self.read_safetensors_metadata(model_path) + log_entries.append("📂 Metadata extraction method: Safetensors") + elif model_path.endswith((".ckpt", ".pth", ".pt", ".bin")): + metadata = self.read_torch_metadata(model_path) + log_entries.append("📂 Metadata extraction method: Checkpoint/Torch") + else: + metadata = {"error": "Unsupported model format"} + log_entries.append("⚠ Unsupported model format detected.") + + log_entries.append(f"✅ Extraction Complete: {self.get_timestamp()}") + + return ("\n".join(log_entries) + "\n\n" + json.dumps(metadata, indent=4),) + + def read_safetensors_metadata(self, model_path): + """Reads metadata from Safetensors models.""" + try: + with safetensors.safe_open(model_path, framework="pt") as f: + return f.metadata() + except Exception as e: + return {"error": f"Safetensors extraction failed: {str(e)}"} + + def read_torch_metadata(self, model_path): + """Reads metadata from Torch-based models.""" + try: + model_data = torch.load(model_path, map_location="cpu") + return {"metadata_keys": list(model_data.keys())} + except Exception as e: + return {"error": f"Torch model extraction failed: {str(e)}"} + + def get_timestamp(self): + """Returns formatted timestamp.""" + return datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S") + +NODE_CLASS_MAPPINGS = { + "ModelMetadataReader": ModelMetadataReader +} diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..93bcf57 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,7 @@ +os +json +torch +safetensors +re +datetime +chardet