diff --git a/.gitignore b/.gitignore index 68bc17f..4c45269 100644 --- a/.gitignore +++ b/.gitignore @@ -1,3 +1,5 @@ +dlib/*.dat + # Byte-compiled / optimized / DLL files __pycache__/ *.py[cod] diff --git a/Inconsolata.otf b/Inconsolata.otf new file mode 100644 index 0000000..3488898 Binary files /dev/null and b/Inconsolata.otf differ diff --git a/README.md b/README.md index 4aeb00d..37c67be 100644 --- a/README.md +++ b/README.md @@ -1,2 +1,23 @@ -# ComfyUI_FaceAnalysis -Extension for ComfyUI to evaluate the similarity between two faces +# Face Analysis for ComfyUI + +This extension uses [DLib](http://dlib.net/) to calculate the Euclidean and Cosine *distance* between two faces. + +Please read the results as follow: + +- **Lower values are better** +- The minimum thresholds are: **EUC 0.6**, **COS 0.07** +- In my tests a value of Euc <0.3 is very good + +## Installation + +Please download the DLIB [Shape Predictor](http://dlib.net/files/shape_predictor_68_face_landmarks.dat.bz2) and the [Face Recognition](http://dlib.net/files/dlib_face_recognition_resnet_model_v1.dat.bz2) models and place them into the `dlib` directory. + +In this repository you also find a workflow that uses IPAdapter to generate a few images and return the distance to the reference fance. + +![face analysis](./face_analysis.jpg) + +## Important notes + +There are many ways to do this. At the moment I'm using DLib as it's fast and easy to use, if there's an actual interest I will release more options (insightface?). + +Also, I'm not an engineer and I don't know what I'm doing, hopefully someone more experienced can chime in. \ No newline at end of file diff --git a/__init__.py b/__init__.py new file mode 100644 index 0000000..3a42eba --- /dev/null +++ b/__init__.py @@ -0,0 +1,3 @@ +from .faceanalysis import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS + +__all__ = ['NODE_CLASS_MAPPINGS', 'NODE_DISPLAY_NAME_MAPPINGS'] diff --git a/dlib/dlib_models_here.txt b/dlib/dlib_models_here.txt new file mode 100644 index 0000000..e69de29 diff --git a/face_analysis.jpg b/face_analysis.jpg new file mode 100644 index 0000000..a0e2a0d Binary files /dev/null and b/face_analysis.jpg differ diff --git a/face_analysis.json b/face_analysis.json new file mode 100644 index 0000000..b7b755e --- /dev/null +++ b/face_analysis.json @@ -0,0 +1,1055 @@ +{ + "last_node_id": 61, + "last_link_id": 159, + "nodes": [ + { + "id": 31, + "type": "CLIPVisionLoader", + "pos": [ + 691, + 172 + ], + "size": { + "0": 290, + "1": 60 + }, + "flags": {}, + "order": 0, + "mode": 0, + "outputs": [ + { + "name": "CLIP_VISION", + "type": "CLIP_VISION", + "links": [ + 82, + 96 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "CLIPVisionLoader" + }, + "widgets_values": [ + "IPAdapter_image_encoder_sd15.safetensors" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 32, + "type": "InsightFaceLoader", + "pos": [ + 691, + 291 + ], + "size": { + "0": 290, + "1": 60 + }, + "flags": {}, + "order": 1, + "mode": 0, + "outputs": [ + { + "name": "INSIGHTFACE", + "type": "INSIGHTFACE", + "links": [ + 84 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "InsightFaceLoader" + }, + "widgets_values": [ + "CPU" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 47, + "type": "ImageCrop+", + "pos": [ + 691, + -198 + ], + "size": { + "0": 315, + "1": 194 + }, + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 115 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 117 + ], + "shape": 3, + "slot_index": 0 + }, + { + "name": "x", + "type": "INT", + "links": null, + "shape": 3 + }, + { + "name": "y", + "type": "INT", + "links": null, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "ImageCrop+" + }, + "widgets_values": [ + 432, + 432, + "top-center", + 0, + 156 + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 37, + "type": "IPAdapterApply", + "pos": [ + 1506, + 120 + ], + "size": { + "0": 315, + "1": 258 + }, + "flags": {}, + "order": 15, + "mode": 0, + "inputs": [ + { + "name": "ipadapter", + "type": "IPADAPTER", + "link": 99, + "slot_index": 0 + }, + { + "name": "clip_vision", + "type": "CLIP_VISION", + "link": 96 + }, + { + "name": "image", + "type": "IMAGE", + "link": 150 + }, + { + "name": "model", + "type": "MODEL", + "link": 94 + }, + { + "name": "attn_mask", + "type": "MASK", + "link": null + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 98 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "IPAdapterApply" + }, + "widgets_values": [ + 0.4, + 0, + "original", + 0, + 1, + false + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 5, + "type": "CLIPTextEncode", + "pos": [ + 657, + 822 + ], + "size": { + "0": 400, + "1": 160 + }, + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 146 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 107 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "CLIPTextEncode" + }, + "widgets_values": [ + "blurry, malformed, distorted, naked, bad eyes, ill" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 4, + "type": "CLIPTextEncode", + "pos": [ + 657, + 592 + ], + "size": { + "0": 400, + "1": 160 + }, + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 148 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 106 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "CLIPTextEncode" + }, + "widgets_values": [ + "closeup photo of a woman wearing a white spring dress in a garden\n\nhigh quality, diffuse light, highly detailed, 4k" + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 9, + "type": "IPAdapterModelLoader", + "pos": [ + 691, + 58 + ], + "size": { + "0": 290, + "1": 60 + }, + "flags": {}, + "order": 2, + "mode": 0, + "outputs": [ + { + "name": "IPADAPTER", + "type": "IPADAPTER", + "links": [ + 81 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "IPAdapterModelLoader" + }, + "widgets_values": [ + "ip-adapter-faceid-plusv2_sd15.bin" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 14, + "type": "LoraLoaderModelOnly", + "pos": [ + 689, + 411 + ], + "size": { + "0": 296.6805419921875, + "1": 82 + }, + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 141 + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 144 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "LoraLoaderModelOnly" + }, + "widgets_values": [ + "ip-adapter-faceid-plusv2_sd15_lora.safetensors", + 0.65 + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 2, + "type": "CheckpointLoaderSimple", + "pos": [ + 198, + 570 + ], + "size": { + "0": 290, + "1": 100 + }, + "flags": {}, + "order": 3, + "mode": 0, + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 141 + ], + "slot_index": 0 + }, + { + "name": "CLIP", + "type": "CLIP", + "links": [ + 146, + 148 + ], + "slot_index": 1 + }, + { + "name": "VAE", + "type": "VAE", + "links": [], + "slot_index": 2 + } + ], + "properties": { + "Node name for S&R": "CheckpointLoaderSimple" + }, + "widgets_values": [ + "sd15/realisticVisionV51_v51VAE.safetensors" + ] + }, + { + "id": 7, + "type": "VAELoader", + "pos": [ + 1918, + 689 + ], + "size": { + "0": 240, + "1": 60 + }, + "flags": {}, + "order": 4, + "mode": 0, + "outputs": [ + { + "name": "VAE", + "type": "VAE", + "links": [ + 8 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "VAELoader" + }, + "widgets_values": [ + "vae-ft-mse-840000-ema-pruned.safetensors" + ] + }, + { + "id": 1, + "type": "KSampler", + "pos": [ + 1931, + 341 + ], + "size": { + "0": 240, + "1": 262 + }, + "flags": {}, + "order": 16, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 98 + }, + { + "name": "positive", + "type": "CONDITIONING", + "link": 106 + }, + { + "name": "negative", + "type": "CONDITIONING", + "link": 107 + }, + { + "name": "latent_image", + "type": "LATENT", + "link": 4 + } + ], + "outputs": [ + { + "name": "LATENT", + "type": "LATENT", + "links": [ + 7 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "KSampler" + }, + "widgets_values": [ + 0, + "fixed", + 35, + 7, + "dpmpp_2m", + "karras", + 1 + ] + }, + { + "id": 38, + "type": "IPAdapterModelLoader", + "pos": [ + 1107, + 32 + ], + "size": { + "0": 315, + "1": 58 + }, + "flags": {}, + "order": 5, + "mode": 0, + "outputs": [ + { + "name": "IPADAPTER", + "type": "IPADAPTER", + "links": [ + 99 + ], + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "IPAdapterModelLoader" + }, + "widgets_values": [ + "ip-adapter-plus-face_sd15.safetensors" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 45, + "type": "PrepImageForClipVision", + "pos": [ + 1081, + -192 + ], + "size": { + "0": 315, + "1": 106 + }, + "flags": {}, + "order": 14, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 117 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 150 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "PrepImageForClipVision" + }, + "widgets_values": [ + "LANCZOS", + "top", + 0.15 + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 3, + "type": "EmptyLatentImage", + "pos": [ + 1629, + 606 + ], + "size": { + "0": 210, + "1": 110 + }, + "flags": {}, + "order": 6, + "mode": 0, + "outputs": [ + { + "name": "LATENT", + "type": "LATENT", + "links": [ + 4 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "EmptyLatentImage" + }, + "widgets_values": [ + 512, + 512, + 4 + ] + }, + { + "id": 6, + "type": "VAEDecode", + "pos": [ + 2222, + 355 + ], + "size": { + "0": 140, + "1": 50 + }, + "flags": {}, + "order": 17, + "mode": 0, + "inputs": [ + { + "name": "samples", + "type": "LATENT", + "link": 7 + }, + { + "name": "vae", + "type": "VAE", + "link": 8 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 159 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "VAEDecode" + } + }, + { + "id": 59, + "type": "FaceAnalysisModels", + "pos": [ + 2140, + 193 + ], + "size": { + "0": 210, + "1": 26 + }, + "flags": {}, + "order": 7, + "mode": 0, + "outputs": [ + { + "name": "ANALYSIS_MODELS", + "type": "ANALYSIS_MODELS", + "links": [ + 155 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "FaceAnalysisModels" + } + }, + { + "id": 61, + "type": "FaceEmbedDistance", + "pos": [ + 2413, + 241 + ], + "size": { + "0": 267, + "1": 66 + }, + "flags": {}, + "order": 18, + "mode": 0, + "inputs": [ + { + "name": "analysis_models", + "type": "ANALYSIS_MODELS", + "link": 155 + }, + { + "name": "reference", + "type": "IMAGE", + "link": 158 + }, + { + "name": "image", + "type": "IMAGE", + "link": 159 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 157 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "FaceEmbedDistance" + } + }, + { + "id": 41, + "type": "PreviewImage", + "pos": [ + 2415, + 356 + ], + "size": [ + 1073.4365928515608, + 1065.4424996423327 + ], + "flags": {}, + "order": 19, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 157 + } + ], + "properties": { + "Node name for S&R": "PreviewImage" + } + }, + { + "id": 10, + "type": "LoadImage", + "pos": [ + 224, + -240 + ], + "size": { + "0": 390.733154296875, + "1": 482.0174560546875 + }, + "flags": {}, + "order": 8, + "mode": 0, + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 115, + 134, + 158 + ], + "shape": 3, + "slot_index": 0 + }, + { + "name": "MASK", + "type": "MASK", + "links": null, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "face4.jpg", + "image" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 33, + "type": "IPAdapterApplyFaceID", + "pos": [ + 1119, + 180 + ], + "size": { + "0": 315, + "1": 326 + }, + "flags": {}, + "order": 13, + "mode": 0, + "inputs": [ + { + "name": "ipadapter", + "type": "IPADAPTER", + "link": 81 + }, + { + "name": "clip_vision", + "type": "CLIP_VISION", + "link": 82 + }, + { + "name": "insightface", + "type": "INSIGHTFACE", + "link": 84 + }, + { + "name": "image", + "type": "IMAGE", + "link": 134 + }, + { + "name": "model", + "type": "MODEL", + "link": 144 + }, + { + "name": "attn_mask", + "type": "MASK", + "link": null + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 94 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "IPAdapterApplyFaceID" + }, + "widgets_values": [ + 0.8, + 0, + "original", + 0, + 1, + true, + 2, + false + ], + "color": "#223", + "bgcolor": "#335" + } + ], + "links": [ + [ + 4, + 3, + 0, + 1, + 3, + "LATENT" + ], + [ + 7, + 1, + 0, + 6, + 0, + "LATENT" + ], + [ + 8, + 7, + 0, + 6, + 1, + "VAE" + ], + [ + 81, + 9, + 0, + 33, + 0, + "IPADAPTER" + ], + [ + 82, + 31, + 0, + 33, + 1, + "CLIP_VISION" + ], + [ + 84, + 32, + 0, + 33, + 2, + "INSIGHTFACE" + ], + [ + 94, + 33, + 0, + 37, + 3, + "MODEL" + ], + [ + 96, + 31, + 0, + 37, + 1, + "CLIP_VISION" + ], + [ + 98, + 37, + 0, + 1, + 0, + "MODEL" + ], + [ + 99, + 38, + 0, + 37, + 0, + "IPADAPTER" + ], + [ + 106, + 4, + 0, + 1, + 1, + "CONDITIONING" + ], + [ + 107, + 5, + 0, + 1, + 2, + "CONDITIONING" + ], + [ + 115, + 10, + 0, + 47, + 0, + "IMAGE" + ], + [ + 117, + 47, + 0, + 45, + 0, + "IMAGE" + ], + [ + 134, + 10, + 0, + 33, + 3, + "IMAGE" + ], + [ + 141, + 2, + 0, + 14, + 0, + "MODEL" + ], + [ + 144, + 14, + 0, + 33, + 4, + "MODEL" + ], + [ + 146, + 2, + 1, + 5, + 0, + "CLIP" + ], + [ + 148, + 2, + 1, + 4, + 0, + "CLIP" + ], + [ + 150, + 45, + 0, + 37, + 2, + "IMAGE" + ], + [ + 155, + 59, + 0, + 61, + 0, + "ANALYSIS_MODELS" + ], + [ + 157, + 61, + 0, + 41, + 0, + "IMAGE" + ], + [ + 158, + 10, + 0, + 61, + 1, + "IMAGE" + ], + [ + 159, + 6, + 0, + 61, + 2, + "IMAGE" + ] + ], + "groups": [], + "config": {}, + "extra": {}, + "version": 0.4 +} \ No newline at end of file diff --git a/faceanalysis.py b/faceanalysis.py new file mode 100644 index 0000000..d4e0d62 --- /dev/null +++ b/faceanalysis.py @@ -0,0 +1,108 @@ +import dlib +import torch +import torchvision.transforms.v2 as T +import os +import numpy as np +from PIL import Image, ImageDraw, ImageFont, ImageColor + +DLIB_DIR = os.path.join(os.path.dirname(os.path.realpath(__file__)), "dlib") + +class FaceAnalysisModels: + @classmethod + def INPUT_TYPES(s): + return {"required": {}} + + RETURN_TYPES = ("ANALYSIS_MODELS", ) + FUNCTION = "load_models" + CATEGORY = "FaceAnalysis" + + def load_models(self): + return ({ + "detector": dlib.get_frontal_face_detector(), + "shape_predict": dlib.shape_predictor(os.path.join(DLIB_DIR, "shape_predictor_68_face_landmarks.dat")), + "face_recog": dlib.face_recognition_model_v1(os.path.join(DLIB_DIR, "dlib_face_recognition_resnet_model_v1.dat")), + }, ) + +class FaceEmbedDistance: + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "analysis_models": ("ANALYSIS_MODELS", ), + "reference": ("IMAGE", ), + "image": ("IMAGE", ), + }, + } + + RETURN_TYPES = ("IMAGE", ) + OUTPUT_NODE = True + FUNCTION = "analize" + CATEGORY = "FaceAnalysis" + + def analize(self, analysis_models, reference, image): + font = ImageFont.truetype(os.path.join(os.path.dirname(os.path.realpath(__file__)), "Inconsolata.otf"), 32) + background_color = ImageColor.getrgb("#000000AA") + txt_height = font.getmask("Q").getbbox()[3] + font.getmetrics()[1] + + self.detector = analysis_models.get("detector") + self.shape_predict = analysis_models.get("shape_predict") + self.face_recog = analysis_models.get("face_recog") + + ref = np.array(T.ToPILImage()(reference[0].permute(2, 0, 1)).convert('RGB')) + ref = self.get_descriptor(ref) + if ref is None: + raise Exception('No face detected in reference image') + + out = [] + + for i in image: + img = np.array(T.ToPILImage()(i.permute(2, 0, 1)).convert('RGB')) + + img = self.get_descriptor(img) + + if img is None: # No face detected + eucl_dist = 1.0 + cos_distance = 1.0 + else: + if ref == img: # Same face + eucl_dist = 0.0 + cos_distance = 0.0 + else: + eucl_dist = np.linalg.norm(np.array(ref) - np.array(img)) + cos_distance = 1 - np.dot(ref, img) / (np.linalg.norm(ref) * np.linalg.norm(img)) + + print(f"\033[96mFace Analysis: Euclidean: {eucl_dist}, Cosine: {cos_distance}\033[0m") + + eucl_dist = round(eucl_dist, 3) + cos_distance = round(cos_distance, 3) + + tmp = T.ToPILImage()(i.permute(2, 0, 1)).convert('RGBA') + txt = Image.new('RGBA', (image.shape[2], txt_height), color=background_color) + draw = ImageDraw.Draw(txt) + draw.text((0, 0), f"EUC: {eucl_dist} | COS-1: {cos_distance}", font=font, fill=(255, 255, 255, 255)) + composite = Image.new('RGBA', tmp.size) + composite.paste(txt, (0, tmp.height - txt.height)) + composite = Image.alpha_composite(tmp, composite) + out.append(T.ToTensor()(composite)) + + img = torch.stack(out).permute(0, 2, 3, 1) + + return(img, ) + + def get_descriptor(self, image): + faces = self.detector(image) + if len(faces) > 0: + shape = self.shape_predict(image, faces[0]) + return self.face_recog.compute_face_descriptor(image, shape) + + return None + +NODE_CLASS_MAPPINGS = { + "FaceEmbedDistance": FaceEmbedDistance, + "FaceAnalysisModels": FaceAnalysisModels, +} + +NODE_DISPLAY_NAME_MAPPINGS = { + "FaceEmbedDistance": "Face Embeds Distance", + "FaceAnalysisModels": "Face Analysis Models", +} diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..a86b17e --- /dev/null +++ b/requirements.txt @@ -0,0 +1 @@ +dlib