From 93059f0fe4ea3881c0d6ec0e327e5d920554277d Mon Sep 17 00:00:00 2001
From: kijai <40791699+kijai@users.noreply.github.com>
Date: Sun, 14 Dec 2025 19:16:28 +0200
Subject: [PATCH] Init
---
.gitignore | 13 +
NLFPoseExtract/align3d.py | 123 +++
NLFPoseExtract/nlf_render.py | 400 +++++++++
__init__.py | 3 +
.../SCAIL_preprocess_example_01.json | 825 ++++++++++++++++++
nodes.py | 243 ++++++
pose_draw/draw_3d_utils.py | 221 +++++
pose_draw/draw_pose_utils.py | 129 +++
pose_draw/draw_utils.py | 658 ++++++++++++++
pyproject.toml | 15 +
readme.md | 12 +
render_3d/render_cylinder.py | 93 ++
render_3d/taichi_cylinder.py | 207 +++++
requirements.txt | 6 +
vitpose_utils/utils.py | 191 ++++
15 files changed, 3139 insertions(+)
create mode 100644 .gitignore
create mode 100644 NLFPoseExtract/align3d.py
create mode 100644 NLFPoseExtract/nlf_render.py
create mode 100644 __init__.py
create mode 100644 example_workflows/SCAIL_preprocess_example_01.json
create mode 100644 nodes.py
create mode 100644 pose_draw/draw_3d_utils.py
create mode 100644 pose_draw/draw_pose_utils.py
create mode 100644 pose_draw/draw_utils.py
create mode 100644 pyproject.toml
create mode 100644 readme.md
create mode 100644 render_3d/render_cylinder.py
create mode 100644 render_3d/taichi_cylinder.py
create mode 100644 requirements.txt
create mode 100644 vitpose_utils/utils.py
diff --git a/.gitignore b/.gitignore
new file mode 100644
index 0000000..d32eacc
--- /dev/null
+++ b/.gitignore
@@ -0,0 +1,13 @@
+output/
+*__pycache__/
+samples*/
+runs/
+checkpoints/
+master_ip
+logs/
+*.DS_Store
+.idea
+tools/
+.vscode/
+convert_*
+*.pt
\ No newline at end of file
diff --git a/NLFPoseExtract/align3d.py b/NLFPoseExtract/align3d.py
new file mode 100644
index 0000000..20e6e06
--- /dev/null
+++ b/NLFPoseExtract/align3d.py
@@ -0,0 +1,123 @@
+import numpy as np
+from scipy.optimize import minimize
+
+
+def solve_new_camera_params_central(three_d_points, focal_length, imshape, new_2d_points):
+ """
+ Solve for new camera parameters by minimizing the error between the original 2D projection points and the new 2D projection points.
+
+ Args:
+ three_d_points (torch.Tensor): N*3 3D points
+ focal_length (float): Focal length of the original camera
+ imshape (tuple): Image size, e.g., [512, 896]
+ original_2d_points (torch.Tensor): N*2 original 2D projection points
+ new_2d_points (torch.Tensor): N*2 new 2D projection points
+
+ Returns:
+ m, n, p, q: Parameters in the new camera intrinsic matrix
+ """
+
+
+ # Objective function: minimize the error between the original projection points and the new projection points
+ def objective(params):
+ m, s, p, q = params
+ # Construct the new camera intrinsic matrix
+ K_new = np.array([
+ [focal_length * m , 0, imshape[1] / 2 + p],
+ [0, focal_length * m * s, imshape[0] / 2 + q],
+ [0, 0, 1]
+ ])
+
+ # Compute the new 2D projection points
+ new_projections = []
+ for point in three_d_points:
+ X, Y, Z = point
+ u = (K_new[0, 0] * X / Z) + K_new[0, 2]
+ v = (K_new[1, 1] * Y / Z) + K_new[1, 2]
+ new_projections.append([u, v])
+ new_projections = np.array(new_projections)
+
+ # Calculate the error between the original 2D projection points and the new projection points
+ # Special handling for the 0th projection point
+ error0 = np.sum((new_2d_points[:1] - new_projections[:1]) ** 2)
+ error = np.sum((new_2d_points[1:] - new_projections[1:]) ** 2)
+ return error0 * 8 + error
+
+ # Initialize parameters m, beta, p, q
+ initial_params = [1.0, 1.0, 0.0, 0.0] # Initial values
+
+ # Use least squares to solve for p, q
+ result = minimize(objective, initial_params, bounds=[(0.7, 1.4), (0.8, 1.15), (-imshape[1], imshape[1]), (-imshape[0], imshape[0])])
+
+ # Output the solution result
+ m, s, p, q = result.x
+ print(f"debug: solved camera params m={m}, s={s}, p={p}, q={q}")
+
+ K_final = np.array([
+ [focal_length * m, 0, imshape[1] / 2 + p],
+ [0, focal_length * m * s, imshape[0] / 2 + q],
+ [0, 0, 1]
+ ])
+
+
+ return K_final, m
+
+
+def solve_new_camera_params_down(three_d_points, focal_length, imshape, new_2d_points):
+ """
+ Solve for new camera parameters by minimizing the error between the original 2D projection points and the new 2D projection points.
+
+ Args:
+ three_d_points (torch.Tensor): N*3 3D points
+ focal_length (float): Focal length of the original camera
+ imshape (tuple): Image size, e.g., [512, 896]
+ original_2d_points (torch.Tensor): N*2 original 2D projection points
+ new_2d_points (torch.Tensor): N*2 new 2D projection points
+
+ Returns:
+ m, n, p, q: Parameters in the new camera intrinsic matrix
+ """
+
+ # Objective function: minimize the error between the original projection points and the new projection points
+ def objective(params):
+ m, s, p, q = params
+ # Construct the new camera intrinsic matrix
+ K_new = np.array([
+ [focal_length * m , 0, imshape[1] / 2 + p],
+ [0, focal_length * m * s, imshape[0] / 2 + q],
+ [0, 0, 1]
+ ])
+
+ # Compute the new 2D projection points
+ new_projections = []
+ for point in three_d_points:
+ X, Y, Z = point
+ u = (K_new[0, 0] * X / Z) + K_new[0, 2]
+ v = (K_new[1, 1] * Y / Z) + K_new[1, 2]
+ new_projections.append([u, v])
+ new_projections = np.array(new_projections)
+
+ # Calculate the error between the original 2D projection points and the new projection points
+ # Special handling for the 0th projection point
+ error0 = np.sum((new_2d_points[:1] - new_projections[:1]) ** 2)
+ error = np.sum((new_2d_points[1:] - new_projections[1:]) ** 2)
+ return error0 + error * 4
+
+ # Initialize parameters m, beta, p, q
+ initial_params = [1.0, 1.0, 0.0, 0.0] # Initial values
+
+ # Use least squares to solve for p, q
+ result = minimize(objective, initial_params, bounds=[(0.7, 1.4), (0.8, 1.15), (-imshape[1], imshape[1]), (-imshape[0], imshape[0])])
+
+ # Output the solution result
+ m, s, p, q = result.x
+ print(f"debug: solved camera params m={m}, s={s}, p={p}, q={q}")
+
+ K_final = np.array([
+ [focal_length * m, 0, imshape[1] / 2 + p],
+ [0, focal_length * m * s, imshape[0] / 2 + q],
+ [0, 0, 1]
+ ])
+
+
+ return K_final, m
diff --git a/NLFPoseExtract/nlf_render.py b/NLFPoseExtract/nlf_render.py
new file mode 100644
index 0000000..9abb43c
--- /dev/null
+++ b/NLFPoseExtract/nlf_render.py
@@ -0,0 +1,400 @@
+import numpy as np
+import torch
+import os, platform, copy
+
+if platform.system() == 'Linux':
+ if 'PYOPENGL_PLATFORM' not in os.environ:
+ os.environ['PYOPENGL_PLATFORM'] = 'egl'
+elif platform.system() == 'Windows':
+ os.environ.pop('PYOPENGL_PLATFORM', None)
+
+from ..render_3d.taichi_cylinder import render_whole
+from ..pose_draw.draw_pose_utils import draw_pose_to_canvas_np
+
+def p3d_single_p2d(points, intrinsic_matrix):
+ X, Y, Z = points[0], points[1], points[2]
+ u = (intrinsic_matrix[0, 0] * X / Z) + intrinsic_matrix[0, 2]
+ v = (intrinsic_matrix[1, 1] * Y / Z) + intrinsic_matrix[1, 2]
+ u_np = u.cpu().numpy()
+ v_np = v.cpu().numpy()
+ return np.array([u_np, v_np])
+
+def process_data_to_COCO_format(joints):
+ """Args:
+ joints: numpy array of shape (24, 2) or (24, 3)
+ Returns:
+ new_joints: numpy array of shape (17, 2) or (17, 3)
+ """
+ if joints.ndim != 2:
+ raise ValueError(f"Expected shape (24,2) or (24,3), got {joints.shape}")
+
+ dim = joints.shape[1] # 2D or 3D
+
+ mapping = {
+ 15: 0, # head
+ 12: 1, # neck
+ 17: 2, # left shoulder
+ 16: 5, # right shoulder
+ 19: 3, # left elbow
+ 18: 6, # right elbow
+ 21: 4, # left hand
+ 20: 7, # right hand
+ 2: 8, # left pelvis
+ 1: 11, # right pelvis
+ 5: 9, # left knee
+ 4: 12, # right knee
+ 8: 10, # left feet
+ 7: 13, # right feet
+ }
+
+ new_joints = np.zeros((18, dim), dtype=joints.dtype)
+ for src, dst in mapping.items():
+ new_joints[dst] = joints[src]
+
+ return new_joints
+
+def intrinsic_matrix_from_field_of_view(imshape, fov_degrees:float =55): # nlf default fov_degrees 55
+ imshape = np.array(imshape)
+ fov_radians = fov_degrees * np.array(np.pi / 180)
+ larger_side = np.max(imshape)
+ focal_length = larger_side / (np.tan(fov_radians / 2) * 2)
+ # intrinsic_matrix 3*3
+ return np.array([
+ [focal_length, 0, imshape[1] / 2],
+ [0, focal_length, imshape[0] / 2],
+ [0, 0, 1],
+ ])
+
+def shift_dwpose_according_to_nlf(smpl_poses, aligned_poses, ori_intrinstics, modified_intrinstics, height, width):
+ ########## warning: 会改变body; shift 之后 body是不准的 ##########
+ for i in range(len(smpl_poses)):
+ persons_joints_list = smpl_poses[i]
+ poses_list = aligned_poses[i]
+ # 对里面每一个人,取关节并进行变形;并且修改2d;如果3d不存在,把2d的手/脸也去掉
+ for person_idx, person_joints in enumerate(persons_joints_list):
+ face = poses_list["faces"][person_idx]
+ right_hand = poses_list["hands"][2 * person_idx]
+ left_hand = poses_list["hands"][2 * person_idx + 1]
+ candidate = poses_list["bodies"]["candidate"][person_idx]
+ # 注意,这里不是coco format
+ person_joint_15_2d_shift = p3d_single_p2d(person_joints[15], modified_intrinstics) - p3d_single_p2d(person_joints[15], ori_intrinstics) if person_joints[15, 2] > 0.01 else np.array([0.0, 0.0]) # face
+ person_joint_20_2d_shift = p3d_single_p2d(person_joints[20], modified_intrinstics) - p3d_single_p2d(person_joints[20], ori_intrinstics) if person_joints[20, 2] > 0.01 else np.array([0.0, 0.0]) # right hand
+ person_joint_21_2d_shift = p3d_single_p2d(person_joints[21], modified_intrinstics) - p3d_single_p2d(person_joints[21], ori_intrinstics) if person_joints[21, 2] > 0.01 else np.array([0.0, 0.0]) # left hand
+
+ face[:, 0] += person_joint_15_2d_shift[0] / width
+ face[:, 1] += person_joint_15_2d_shift[1] / height
+ right_hand[:, 0] += person_joint_20_2d_shift[0] / width
+ right_hand[:, 1] += person_joint_20_2d_shift[1] / height
+ left_hand[:, 0] += person_joint_21_2d_shift[0] / width
+ left_hand[:, 1] += person_joint_21_2d_shift[1] / height
+ candidate[:, 0] += person_joint_15_2d_shift[0] / width
+ candidate[:, 1] += person_joint_15_2d_shift[1] / height
+
+
+def get_single_pose_cylinder_specs(args):
+ """Helper function for rendering a single pose, used for parallel processing."""
+ idx, pose, focal, princpt, height, width, colors, limb_seq, draw_seq = args
+ cylinder_specs = []
+
+ for joints3d in pose: # 多人
+ joints3d = joints3d.cpu().numpy()
+ joints3d = process_data_to_COCO_format(joints3d)
+ for line_idx in draw_seq:
+ line = limb_seq[line_idx]
+ start, end = line[0], line[1]
+ if np.sum(joints3d[start]) == 0 or np.sum(joints3d[end]) == 0:
+ continue
+ else:
+ cylinder_specs.append((joints3d[start], joints3d[end], colors[line_idx]))
+ return cylinder_specs
+
+
+def collect_smpl_poses(data):
+ uncollected_smpl_poses = [item['nlfpose'] for item in data]
+ smpl_poses = [[] for _ in range(len(uncollected_smpl_poses))]
+ for frame_idx in range(len(uncollected_smpl_poses)):
+ for person_idx in range(len(uncollected_smpl_poses[frame_idx])): # 每个人(每个bbox)只给出一个pose
+ if len(uncollected_smpl_poses[frame_idx][person_idx]) > 0: # 有返回的骨骼
+ smpl_poses[frame_idx].append(uncollected_smpl_poses[frame_idx][person_idx][0])
+ else:
+ smpl_poses[frame_idx].append(torch.zeros((24, 3), dtype=torch.float32)) # 没有检测到人,就放一个全0的
+
+ return smpl_poses
+
+
+
+def collect_smpl_poses_samurai(data):
+ uncollected_smpl_poses = [item['nlfpose'] for item in data]
+ smpl_poses_first = [[] for _ in range(len(uncollected_smpl_poses))]
+ smpl_poses_second = [[] for _ in range(len(uncollected_smpl_poses))]
+
+ for frame_idx in range(len(uncollected_smpl_poses)):
+ for person_idx in range(len(uncollected_smpl_poses[frame_idx])): # 每个人(每个bbox)只给出一个pose
+ if len(uncollected_smpl_poses[frame_idx][person_idx]) > 0: # 有返回的骨骼
+ if person_idx == 0:
+ smpl_poses_first[frame_idx].append(uncollected_smpl_poses[frame_idx][person_idx][0])
+ elif person_idx == 1:
+ smpl_poses_second[frame_idx].append(uncollected_smpl_poses[frame_idx][person_idx][0])
+ else:
+ if person_idx == 0:
+ smpl_poses_first[frame_idx].append(torch.zeros((24, 3), dtype=torch.float32)) # 没有检测到人,就放一个全0的
+ elif person_idx == 1:
+ smpl_poses_second[frame_idx].append(torch.zeros((24, 3), dtype=torch.float32))
+
+ return smpl_poses_first, smpl_poses_second
+
+
+
+
+def render_nlf_as_images(smpl_poses, dw_poses, height, width, video_length, intrinsic_matrix=None, draw_2d=True):
+ """ return a list of images """
+
+ base_colors_255_dict = {
+ # Warm Colors for Right Side (R.) - Red, Orange, Yellow
+ "Red": [255, 0, 0],
+ "Orange": [255, 85, 0],
+ "Golden Orange": [255, 170, 0],
+ "Yellow": [255, 240, 0],
+ "Yellow-Green": [180, 255, 0],
+ # Cool Colors for Left Side (L.) - Green, Blue, Purple
+ "Bright Green": [0, 255, 0],
+ "Light Green-Blue": [0, 255, 85],
+ "Aqua": [0, 255, 170],
+ "Cyan": [0, 255, 255],
+ "Sky Blue": [0, 170, 255],
+ "Medium Blue": [0, 85, 255],
+ "Pure Blue": [0, 0, 255],
+ "Purple-Blue": [85, 0, 255],
+ "Medium Purple": [170, 0, 255],
+ # Neutral/Central Colors (e.g., for Neck, Nose, Eyes, Ears)
+ "Grey": [150, 150, 150],
+ "Pink-Magenta": [255, 0, 170],
+ "Dark Pink": [255, 0, 85],
+ "Violet": [100, 0, 255],
+ "Dark Violet": [50, 0, 255],
+ }
+
+ ordered_colors_255 = [
+ base_colors_255_dict["Red"], # Neck -> R. Shoulder (Red)
+ base_colors_255_dict["Cyan"], # Neck -> L. Shoulder (Cyan)
+ base_colors_255_dict["Orange"], # R. Shoulder -> R. Elbow (Orange)
+ base_colors_255_dict["Golden Orange"], # R. Elbow -> R. Wrist (Golden Orange)
+ base_colors_255_dict["Sky Blue"], # L. Shoulder -> L. Elbow (Sky Blue)
+ base_colors_255_dict["Medium Blue"], # L. Elbow -> L. Wrist (Medium Blue)
+ base_colors_255_dict["Yellow-Green"], # Neck -> R. Hip ( Yellow-Green)
+ base_colors_255_dict["Bright Green"], # R. Hip -> R. Knee (Bright Green - transitioning warm to cool spectrum)
+ base_colors_255_dict["Light Green-Blue"], # R. Knee -> R. Ankle (Light Green-Blue - transitioning)
+ base_colors_255_dict["Pure Blue"], # Neck -> L. Hip (Pure Blue)
+ base_colors_255_dict["Purple-Blue"], # L. Hip -> L. Knee (Purple-Blue)
+ base_colors_255_dict["Medium Purple"], # L. Knee -> L. Ankle (Medium Purple)
+ base_colors_255_dict["Grey"], # Neck -> Nose (Grey)
+ base_colors_255_dict["Pink-Magenta"], # Nose -> R. Eye (Pink/Magenta)
+ base_colors_255_dict["Dark Violet"], # R. Eye -> R. Ear (Dark Pink)
+ base_colors_255_dict["Pink-Magenta"], # Nose -> L. Eye (Violet)
+ base_colors_255_dict["Dark Violet"], # L. Eye -> L. Ear (Dark Violet)
+ ]
+
+ limb_seq = [
+ [1, 2], # 0 Neck -> R. Shoulder
+ [1, 5], # 1 Neck -> L. Shoulder
+ [2, 3], # 2 R. Shoulder -> R. Elbow
+ [3, 4], # 3 R. Elbow -> R. Wrist
+ [5, 6], # 4 L. Shoulder -> L. Elbow
+ [6, 7], # 5 L. Elbow -> L. Wrist
+ [1, 8], # 6 Neck -> R. Hip
+ [8, 9], # 7 R. Hip -> R. Knee
+ [9, 10], # 8 R. Knee -> R. Ankle
+ [1, 11], # 9 Neck -> L. Hip
+ [11, 12], # 10 L. Hip -> L. Knee
+ [12, 13], # 11 L. Knee -> L. Ankle
+ [1, 0], # 12 Neck -> Nose
+ [0, 14], # 13 Nose -> R. Eye
+ [14, 16], # 14 R. Eye -> R. Ear
+ [0, 15], # 15 Nose -> L. Eye
+ [15, 17], # 16 L. Eye -> L. Ear
+ ]
+
+ draw_seq = [0, 2, 3, # Neck -> R. Shoulder -> R. Elbow -> R. Wrist
+ 1, 4, 5, # Neck -> L. Shoulder -> L. Elbow -> L. Wrist
+ 6, 7, 8, # Neck -> R. Hip -> R. Knee -> R. Ankle
+ 9, 10, 11, # Neck -> L. Hip -> L. Knee -> L. Ankle
+ 12, # Neck -> Nose
+ 13, 14, # Nose -> R. Eye -> R. Ear
+ 15, 16, # Nose -> L. Eye -> L. Ear
+ ] # Expanding outward from the proximal end
+
+ colors = [[c / 300 + 0.15 for c in color_rgb] + [0.8] for color_rgb in ordered_colors_255]
+
+ if dw_poses is not None:
+ aligned_poses = copy.deepcopy(dw_poses)
+
+ if intrinsic_matrix is None:
+ intrinsic_matrix = intrinsic_matrix_from_field_of_view((height, width))
+ focal_x = intrinsic_matrix[0,0]
+ focal_y = intrinsic_matrix[1,1]
+ princpt = (intrinsic_matrix[0,2], intrinsic_matrix[1,2]) # (cx, cy)
+
+
+ # obtain cylinder_specs for each frame
+ cylinder_specs_list = []
+ for i in range(video_length):
+ cylinder_specs = get_single_pose_cylinder_specs((i, smpl_poses[i], None, None, None, None, colors, limb_seq, draw_seq))
+ cylinder_specs_list.append(cylinder_specs)
+
+
+ frames_np_rgba = render_whole(cylinder_specs_list, H=height, W=width, fx=focal_x, fy=focal_y, cx=princpt[0], cy=princpt[1])
+ if dw_poses is not None and draw_2d:
+ canvas_2d = draw_pose_to_canvas_np(aligned_poses, pool=None, H=height, W=width, reshape_scale=0, show_feet_flag=False, show_body_flag=False, show_cheek_flag=True, dw_hand=True)
+
+ for i in range(len(frames_np_rgba)):
+ frame_img = frames_np_rgba[i]
+ canvas_img = canvas_2d[i]
+ mask = canvas_img != 0
+ frame_img[:, :, :3][mask] = canvas_img[mask]
+ frames_np_rgba[i] = frame_img
+
+ return frames_np_rgba
+
+
+def render_multi_nlf_as_images(data, dw_poses, intrinsic_matrix=None, draw_2d=True):
+ """ return a list of images """
+ height, width = data[0]['video_height'], data[0]['video_width']
+ video_length = len(data)
+
+ second_person_base_colors_255_dict = {
+ # Warm Colors for Right Side (R.) - Red, Orange, Yellow
+ "Red": [255, 20, 20],
+ "Orange": [255, 60, 0],
+ "Golden Orange": [255, 110, 0],
+ "Yellow": [255, 200, 0],
+ "Yellow-Green": [160, 255, 40],
+
+ # Cool Colors for Left Side (L.) - Green, Blue, Purple
+ "Bright Green": [0, 255, 50],
+ "Light Green-Blue": [0, 255, 100],
+ "Aqua": [0, 255, 200],
+ "Cyan": [0, 230, 255],
+ "Sky Blue": [0, 130, 255],
+ "Medium Blue": [0, 70, 255],
+ "Pure Blue": [0, 0, 255],
+ "Purple-Blue": [80, 0, 255],
+ "Medium Purple": [160, 0, 255],
+
+ # Neutral/Central Colors (e.g., for Neck, Nose, Eyes, Ears)
+ "Grey": [130, 130, 130],
+ "Pink-Magenta": [255, 0, 150],
+ "Dark Pink": [255, 0, 100],
+ "Violet": [120, 0, 255],
+ "Dark Violet": [60, 0, 255],
+ }
+
+ first_person_base_colors_255_dict = {
+ # Warm Colors for Right Side (R.) - Red, Orange, Yellow
+ "Red": [255, 150, 150],
+ "Orange": [255, 180, 140],
+ "Golden Orange": [255, 215, 150],
+ "Yellow": [255, 240, 170],
+ "Yellow-Green": [200, 255, 100],
+
+ # Cool Colors for Left Side (L.) - Green, Blue, Purple
+ "Bright Green": [100, 255, 100],
+ "Light Green-Blue": [140, 255, 180],
+ "Aqua": [150, 240, 200],
+ "Cyan": [180, 230, 240],
+ "Sky Blue": [160, 200, 255],
+ "Medium Blue": [100, 120, 255],
+ "Pure Blue": [120, 140, 255],
+ "Purple-Blue": [180, 90, 255],
+ "Medium Purple": [190, 120, 255],
+
+ # Neutral/Central Colors (e.g., for Neck, Nose, Eyes, Ears)
+ "Grey": [210, 210, 210],
+ "Pink-Magenta": [255, 120, 200],
+ "Dark Pink": [255, 150, 180],
+ "Violet": [200, 90, 255],
+ "Dark Violet": [130, 80, 255],
+ }
+
+ base_colors_255_dict_list = [first_person_base_colors_255_dict, second_person_base_colors_255_dict]
+ ordered_colors_255_list = [[
+ base_colors_255_dict["Red"], # Neck -> R. Shoulder (Red)
+ base_colors_255_dict["Cyan"], # Neck -> L. Shoulder (Cyan)
+ base_colors_255_dict["Orange"], # R. Shoulder -> R. Elbow (Orange)
+ base_colors_255_dict["Golden Orange"], # R. Elbow -> R. Wrist (Golden Orange)
+ base_colors_255_dict["Sky Blue"], # L. Shoulder -> L. Elbow (Sky Blue)
+ base_colors_255_dict["Medium Blue"], # L. Elbow -> L. Wrist (Medium Blue)
+ base_colors_255_dict["Yellow-Green"], # Neck -> R. Hip ( Yellow-Green)
+ base_colors_255_dict["Bright Green"], # R. Hip -> R. Knee (Bright Green - transitioning warm to cool spectrum)
+ base_colors_255_dict["Light Green-Blue"], # R. Knee -> R. Ankle (Light Green-Blue - transitioning)
+ base_colors_255_dict["Pure Blue"], # Neck -> L. Hip (Pure Blue)
+ base_colors_255_dict["Purple-Blue"], # L. Hip -> L. Knee (Purple-Blue)
+ base_colors_255_dict["Medium Purple"], # L. Knee -> L. Ankle (Medium Purple)
+ base_colors_255_dict["Grey"], # Neck -> Nose (Grey)
+ base_colors_255_dict["Pink-Magenta"], # Nose -> R. Eye (Pink/Magenta)
+ base_colors_255_dict["Dark Violet"], # R. Eye -> R. Ear (Dark Pink)
+ base_colors_255_dict["Pink-Magenta"], # Nose -> L. Eye (Violet)
+ base_colors_255_dict["Dark Violet"], # L. Eye -> L. Ear (Dark Violet)
+ ] for base_colors_255_dict in base_colors_255_dict_list]
+
+ limb_seq = [
+ [1, 2], # 0 Neck -> R. Shoulder
+ [1, 5], # 1 Neck -> L. Shoulder
+ [2, 3], # 2 R. Shoulder -> R. Elbow
+ [3, 4], # 3 R. Elbow -> R. Wrist
+ [5, 6], # 4 L. Shoulder -> L. Elbow
+ [6, 7], # 5 L. Elbow -> L. Wrist
+ [1, 8], # 6 Neck -> R. Hip
+ [8, 9], # 7 R. Hip -> R. Knee
+ [9, 10], # 8 R. Knee -> R. Ankle
+ [1, 11], # 9 Neck -> L. Hip
+ [11, 12], # 10 L. Hip -> L. Knee
+ [12, 13], # 11 L. Knee -> L. Ankle
+ [1, 0], # 12 Neck -> Nose
+ [0, 14], # 13 Nose -> R. Eye
+ [14, 16], # 14 R. Eye -> R. Ear
+ [0, 15], # 15 Nose -> L. Eye
+ [15, 17], # 16 L. Eye -> L. Ear
+ ]
+
+ draw_seq = [0, 2, 3, # Neck -> R. Shoulder -> R. Elbow -> R. Wrist
+ 1, 4, 5, # Neck -> L. Shoulder -> L. Elbow -> L. Wrist
+ 6, 7, 8, # Neck -> R. Hip -> R. Knee -> R. Ankle
+ 9, 10, 11, # Neck -> L. Hip -> L. Knee -> L. Ankle
+ 12, # Neck -> Nose
+ 13, 14, # Nose -> R. Eye -> R. Ear
+ 15, 16, # Nose -> L. Eye -> L. Ear
+ ] # Expanding outward from the proximal end
+
+ colors_first = [[c / 300 + 0.15 for c in color_rgb] + [0.8] for color_rgb in ordered_colors_255_list[0]]
+ colors_second = [[c / 300 + 0.15 for c in color_rgb] + [0.8] for color_rgb in ordered_colors_255_list[1]]
+
+ smpl_poses_first, smpl_poses_second = collect_smpl_poses_samurai(data)
+
+
+ if intrinsic_matrix is None:
+ intrinsic_matrix = intrinsic_matrix_from_field_of_view((height, width))
+ focal_x = intrinsic_matrix[0,0]
+ focal_y = intrinsic_matrix[1,1]
+ princpt = (intrinsic_matrix[0,2], intrinsic_matrix[1,2]) # (cx, cy)
+
+ # obtain cylinder_specs for each frame
+ cylinder_specs_list = []
+ for i in range(video_length):
+ cylinder_specs_first = get_single_pose_cylinder_specs((i, smpl_poses_first[i], None, None, None, None, colors_first, limb_seq, draw_seq))
+ cylinder_specs_second = get_single_pose_cylinder_specs((i, smpl_poses_second[i], None, None, None, None, colors_second, limb_seq, draw_seq))
+ cylinder_specs = cylinder_specs_first + cylinder_specs_second
+ cylinder_specs_list.append(cylinder_specs)
+
+
+ frames_np_rgba = render_whole(cylinder_specs_list, H=height, W=width, fx=focal_x, fy=focal_y, cx=princpt[0], cy=princpt[1])
+ if dw_poses is not None and draw_2d:
+ aligned_poses = copy.deepcopy(dw_poses)
+ canvas_2d = draw_pose_to_canvas_np(aligned_poses, pool=None, H=height, W=width, reshape_scale=0, show_feet_flag=False, show_body_flag=False, show_cheek_flag=True, dw_hand=True)
+ for i in range(len(frames_np_rgba)):
+ frame_img = frames_np_rgba[i]
+ canvas_img = canvas_2d[i]
+ mask = canvas_img != 0
+ frame_img[:, :, :3][mask] = canvas_img[mask]
+ frames_np_rgba[i] = frame_img
+
+ return frames_np_rgba
diff --git a/__init__.py b/__init__.py
new file mode 100644
index 0000000..2e96bd6
--- /dev/null
+++ b/__init__.py
@@ -0,0 +1,3 @@
+from .nodes import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
+
+__all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS"]
\ No newline at end of file
diff --git a/example_workflows/SCAIL_preprocess_example_01.json b/example_workflows/SCAIL_preprocess_example_01.json
new file mode 100644
index 0000000..0b9b4c4
--- /dev/null
+++ b/example_workflows/SCAIL_preprocess_example_01.json
@@ -0,0 +1,825 @@
+{
+ "id": "c6e410bc-5e2c-460b-ae81-c91b6094fbb1",
+ "revision": 0,
+ "last_node_id": 381,
+ "last_link_id": 706,
+ "nodes": [
+ {
+ "id": 379,
+ "type": "VHS_LoadVideo",
+ "pos": [
+ -728.1399554193044,
+ -1903.2296105502905
+ ],
+ "size": [
+ 255.8291015625,
+ 743.5855352640658
+ ],
+ "flags": {},
+ "order": 0,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "meta_batch",
+ "shape": 7,
+ "type": "VHS_BatchManager",
+ "link": null
+ },
+ {
+ "name": "vae",
+ "shape": 7,
+ "type": "VAE",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 684
+ ]
+ },
+ {
+ "name": "frame_count",
+ "type": "INT",
+ "links": []
+ },
+ {
+ "name": "audio",
+ "type": "AUDIO",
+ "links": null
+ },
+ {
+ "name": "video_info",
+ "type": "VHS_VIDEOINFO",
+ "links": null
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-videohelpersuite",
+ "ver": "8550981384301e9bc5bfea83e5c2c75258102593",
+ "Node name for S&R": "VHS_LoadVideo"
+ },
+ "widgets_values": {
+ "video": "vid.mp4",
+ "force_rate": 0,
+ "custom_width": 0,
+ "custom_height": 0,
+ "frame_load_cap": 81,
+ "skip_first_frames": 0,
+ "select_every_nth": 2,
+ "format": "AnimateDiff",
+ "videopreview": {
+ "hidden": false,
+ "paused": false,
+ "params": {
+ "filename": "vid.mp4",
+ "type": "input",
+ "format": "video/mp4",
+ "force_rate": 0,
+ "custom_width": 0,
+ "custom_height": 0,
+ "frame_load_cap": 81,
+ "skip_first_frames": 0,
+ "select_every_nth": 2
+ }
+ }
+ }
+ },
+ {
+ "id": 358,
+ "type": "OnnxDetectionModelLoader",
+ "pos": [
+ -731.2840742077292,
+ -2127.3788215536956
+ ],
+ "size": [
+ 304.0484375,
+ 106
+ ],
+ "flags": {},
+ "order": 1,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "model",
+ "type": "POSEMODEL",
+ "links": [
+ 677,
+ 680
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "ComfyUI-WanAnimatePreprocess",
+ "ver": "65502e208ae89231619e3def41bc8bafe6fc4f1e",
+ "Node name for S&R": "OnnxDetectionModelLoader"
+ },
+ "widgets_values": [
+ "vitpose-l-wholebody.onnx",
+ "onnx\\yolov10m.onnx",
+ "CUDAExecutionProvider"
+ ]
+ },
+ {
+ "id": 361,
+ "type": "NLFPredict",
+ "pos": [
+ 485.2315622592827,
+ -2157.1952980266337
+ ],
+ "size": [
+ 157.564453125,
+ 46
+ ],
+ "flags": {},
+ "order": 7,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "model",
+ "type": "NLFMODEL",
+ "link": 627
+ },
+ {
+ "name": "images",
+ "type": "IMAGE",
+ "link": 701
+ }
+ ],
+ "outputs": [
+ {
+ "name": "pose_results",
+ "type": "NLFPRED",
+ "links": [
+ 656
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "056d8ad96ffa5a223a8cd88c900a573a8d450e22",
+ "Node name for S&R": "NLFPredict"
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 362,
+ "type": "DownloadAndLoadNLFModel",
+ "pos": [
+ -279.5657072597091,
+ -2231.190892106357
+ ],
+ "size": [
+ 606.9488335754912,
+ 82
+ ],
+ "flags": {},
+ "order": 2,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "nlf_model",
+ "type": "NLFMODEL",
+ "links": [
+ 627
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "056d8ad96ffa5a223a8cd88c900a573a8d450e22",
+ "Node name for S&R": "DownloadAndLoadNLFModel"
+ },
+ "widgets_values": [
+ "https://github.com/isarandi/nlf/releases/download/v0.3.2/nlf_l_multi_0.3.2.torchscript",
+ true
+ ]
+ },
+ {
+ "id": 359,
+ "type": "LoadImage",
+ "pos": [
+ -369.63293579008285,
+ -1510.3591727642465
+ ],
+ "size": [
+ 282.798828125,
+ 314
+ ],
+ "flags": {},
+ "order": 3,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 698
+ ]
+ },
+ {
+ "name": "MASK",
+ "type": "MASK",
+ "links": null
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.4.0",
+ "Node name for S&R": "LoadImage"
+ },
+ "widgets_values": [
+ "anya.jpg",
+ "image"
+ ]
+ },
+ {
+ "id": 381,
+ "type": "Reroute",
+ "pos": [
+ 144.49613132898912,
+ -2052.387551799464
+ ],
+ "size": [
+ 75,
+ 26
+ ],
+ "flags": {},
+ "order": 5,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "",
+ "type": "*",
+ "link": 700
+ }
+ ],
+ "outputs": [
+ {
+ "name": "",
+ "type": "IMAGE",
+ "links": [
+ 701,
+ 702
+ ]
+ }
+ ],
+ "properties": {
+ "showOutputText": false,
+ "horizontal": false
+ }
+ },
+ {
+ "id": 380,
+ "type": "VHS_VideoCombine",
+ "pos": [
+ 1061.113369865757,
+ -2147.425691347535
+ ],
+ "size": [
+ 329.65692831178876,
+ 889.8996245456303
+ ],
+ "flags": {},
+ "order": 11,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "images",
+ "type": "IMAGE",
+ "link": 685
+ },
+ {
+ "name": "audio",
+ "shape": 7,
+ "type": "AUDIO",
+ "link": null
+ },
+ {
+ "name": "meta_batch",
+ "shape": 7,
+ "type": "VHS_BatchManager",
+ "link": null
+ },
+ {
+ "name": "vae",
+ "shape": 7,
+ "type": "VAE",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "name": "Filenames",
+ "type": "VHS_FILENAMES",
+ "links": null
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-videohelpersuite",
+ "ver": "8550981384301e9bc5bfea83e5c2c75258102593",
+ "Node name for S&R": "VHS_VideoCombine"
+ },
+ "widgets_values": {
+ "frame_rate": 16,
+ "loop_count": 0,
+ "filename_prefix": "SCAIL_pose",
+ "format": "video/h264-mp4",
+ "pix_fmt": "yuv420p",
+ "crf": 19,
+ "save_metadata": true,
+ "trim_to_audio": false,
+ "pingpong": false,
+ "save_output": false,
+ "videopreview": {
+ "hidden": false,
+ "paused": false,
+ "params": {
+ "filename": "SCAIL_pose_00030.mp4",
+ "subfolder": "",
+ "type": "temp",
+ "format": "video/h264-mp4",
+ "frame_rate": 16,
+ "workflow": "SCAIL_pose_00030.png",
+ "fullpath": "N:\\AI\\ComfyUI\\temp\\SCAIL_pose_00030.mp4"
+ }
+ }
+ }
+ },
+ {
+ "id": 376,
+ "type": "PoseDetectionVitPoseToDWPose",
+ "pos": [
+ 330.07991567045417,
+ -1961.6148428238132
+ ],
+ "size": [
+ 271.9595703125,
+ 46
+ ],
+ "flags": {},
+ "order": 8,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "vitpose_model",
+ "type": "POSEMODEL",
+ "link": 677
+ },
+ {
+ "name": "images",
+ "type": "IMAGE",
+ "link": 702
+ }
+ ],
+ "outputs": [
+ {
+ "name": "dw_poses",
+ "type": "DWPOSES",
+ "links": [
+ 704
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "PoseDetectionVitPoseToDWPose"
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 370,
+ "type": "RenderNLFPoses",
+ "pos": [
+ 747.8706167185883,
+ -2157.040108369948
+ ],
+ "size": [
+ 270,
+ 146
+ ],
+ "flags": {},
+ "order": 10,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "nlf_poses",
+ "type": "NLFPRED",
+ "link": 656
+ },
+ {
+ "name": "dw_poses",
+ "shape": 7,
+ "type": "DWPOSES",
+ "link": 704
+ },
+ {
+ "name": "ref_dw_pose",
+ "shape": 7,
+ "type": "DWPOSES",
+ "link": 706
+ },
+ {
+ "name": "width",
+ "type": "INT",
+ "widget": {
+ "name": "width"
+ },
+ "link": 688
+ },
+ {
+ "name": "height",
+ "type": "INT",
+ "widget": {
+ "name": "height"
+ },
+ "link": 689
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 685
+ ]
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": []
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "RenderNLFPoses"
+ },
+ "widgets_values": [
+ 512,
+ 896
+ ]
+ },
+ {
+ "id": 372,
+ "type": "ImageResizeKJv2",
+ "pos": [
+ -379.6838809890405,
+ -1915.7782190865344
+ ],
+ "size": [
+ 270,
+ 336
+ ],
+ "flags": {},
+ "order": 4,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 684
+ },
+ {
+ "name": "mask",
+ "shape": 7,
+ "type": "MASK",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 700
+ ]
+ },
+ {
+ "name": "width",
+ "type": "INT",
+ "links": [
+ 688,
+ 695
+ ]
+ },
+ {
+ "name": "height",
+ "type": "INT",
+ "links": [
+ 689,
+ 696
+ ]
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": null
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-kjnodes",
+ "ver": "1b9af2322c009e0171f1c7c2f8457ca69afa1ec0",
+ "Node name for S&R": "ImageResizeKJv2"
+ },
+ "widgets_values": [
+ 512,
+ 896,
+ "bilinear",
+ "crop",
+ "0, 0, 0",
+ "center",
+ 2,
+ "cpu",
+ "
| Output: | 81 x 512 x 896 | 425.25MB |
"
+ ]
+ },
+ {
+ "id": 374,
+ "type": "ImageResizeKJv2",
+ "pos": [
+ -46.738037813889264,
+ -1635.669074975477
+ ],
+ "size": [
+ 270,
+ 336
+ ],
+ "flags": {},
+ "order": 6,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 698
+ },
+ {
+ "name": "mask",
+ "shape": 7,
+ "type": "MASK",
+ "link": null
+ },
+ {
+ "name": "width",
+ "type": "INT",
+ "widget": {
+ "name": "width"
+ },
+ "link": 695
+ },
+ {
+ "name": "height",
+ "type": "INT",
+ "widget": {
+ "name": "height"
+ },
+ "link": 696
+ }
+ ],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 682
+ ]
+ },
+ {
+ "name": "width",
+ "type": "INT",
+ "links": null
+ },
+ {
+ "name": "height",
+ "type": "INT",
+ "links": null
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": null
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-kjnodes",
+ "ver": "1b9af2322c009e0171f1c7c2f8457ca69afa1ec0",
+ "Node name for S&R": "ImageResizeKJv2"
+ },
+ "widgets_values": [
+ 512,
+ 896,
+ "bilinear",
+ "pad",
+ "0, 0, 0",
+ "center",
+ 2,
+ "cpu",
+ "| Output: | 1 x 512 x 896 | 5.25MB |
"
+ ]
+ },
+ {
+ "id": 377,
+ "type": "PoseDetectionVitPoseToDWPose",
+ "pos": [
+ 361.20948481109036,
+ -1818.3087110204879
+ ],
+ "size": [
+ 271.9595703125,
+ 46
+ ],
+ "flags": {},
+ "order": 9,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "vitpose_model",
+ "type": "POSEMODEL",
+ "link": 680
+ },
+ {
+ "name": "images",
+ "type": "IMAGE",
+ "link": 682
+ }
+ ],
+ "outputs": [
+ {
+ "name": "dw_poses",
+ "type": "DWPOSES",
+ "links": [
+ 706
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "PoseDetectionVitPoseToDWPose"
+ },
+ "widgets_values": []
+ }
+ ],
+ "links": [
+ [
+ 627,
+ 362,
+ 0,
+ 361,
+ 0,
+ "NLFMODEL"
+ ],
+ [
+ 656,
+ 361,
+ 0,
+ 370,
+ 0,
+ "NLFPRED"
+ ],
+ [
+ 677,
+ 358,
+ 0,
+ 376,
+ 0,
+ "POSEMODEL"
+ ],
+ [
+ 680,
+ 358,
+ 0,
+ 377,
+ 0,
+ "POSEMODEL"
+ ],
+ [
+ 682,
+ 374,
+ 0,
+ 377,
+ 1,
+ "IMAGE"
+ ],
+ [
+ 684,
+ 379,
+ 0,
+ 372,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 685,
+ 370,
+ 0,
+ 380,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 688,
+ 372,
+ 1,
+ 370,
+ 3,
+ "INT"
+ ],
+ [
+ 689,
+ 372,
+ 2,
+ 370,
+ 4,
+ "INT"
+ ],
+ [
+ 695,
+ 372,
+ 1,
+ 374,
+ 2,
+ "INT"
+ ],
+ [
+ 696,
+ 372,
+ 2,
+ 374,
+ 3,
+ "INT"
+ ],
+ [
+ 698,
+ 359,
+ 0,
+ 374,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 700,
+ 372,
+ 0,
+ 381,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 701,
+ 381,
+ 0,
+ 361,
+ 1,
+ "IMAGE"
+ ],
+ [
+ 702,
+ 381,
+ 0,
+ 376,
+ 1,
+ "IMAGE"
+ ],
+ [
+ 704,
+ 376,
+ 0,
+ 370,
+ 1,
+ "DWPOSES"
+ ],
+ [
+ 706,
+ 377,
+ 0,
+ 370,
+ 2,
+ "DWPOSES"
+ ]
+ ],
+ "groups": [],
+ "config": {},
+ "extra": {
+ "ds": {
+ "scale": 0.8140274938684753,
+ "offset": [
+ 1341.9401489123059,
+ 2475.387199967974
+ ]
+ },
+ "frontendVersion": "1.35.3",
+ "workflowRendererVersion": "LG",
+ "node_versions": {
+ "ComfyUI-WanVideoWrapper": "5a2383621a05825d0d0437781afcb8552d9590fd",
+ "comfy-core": "0.3.26",
+ "ComfyUI-VideoHelperSuite": "0a75c7958fe320efcb052f1d9f8451fd20c730a8"
+ },
+ "VHS_latentpreview": true,
+ "VHS_latentpreviewrate": 0,
+ "VHS_MetadataImage": true,
+ "VHS_KeepIntermediate": true
+ },
+ "version": 0.4
+}
\ No newline at end of file
diff --git a/nodes.py b/nodes.py
new file mode 100644
index 0000000..a8b2078
--- /dev/null
+++ b/nodes.py
@@ -0,0 +1,243 @@
+import os
+import torch
+from tqdm import tqdm
+import numpy as np
+import folder_paths
+import cv2
+import logging
+import copy
+script_directory = os.path.dirname(os.path.abspath(__file__))
+
+from comfy import model_management as mm
+from comfy.utils import ProgressBar
+device = mm.get_torch_device()
+offload_device = mm.unet_offload_device()
+
+folder_paths.add_model_folder_path("detection", os.path.join(folder_paths.models_dir, "detection"))
+
+from .vitpose_utils.utils import bbox_from_detector, crop, load_pose_metas_from_kp2ds_seq, aaposemeta_to_dwpose_scail
+
+def scale_faces(poses, pose_2d_ref):
+ # Input: two lists of dict, poses[0]['faces'].shape: 1, 68, 2 , poses_ref[0]['faces'].shape: 1, 68, 2
+ # Scale the facial keypoints in poses according to the center point of the face
+ # That is: calculate the distance from the center point (idx: 30) to other facial keypoints in ref,
+ # and the same for poses, then get scale_n as the ratio
+ # Clamp scale_n to the range 0.8-1.5, then apply it to poses
+ # Note: poses are modified in place
+
+ ref = pose_2d_ref[0]
+ pose_0 = poses[0]
+
+ face_0 = pose_0['faces'] # shape: (1, 68, 2)
+ face_ref = ref['faces']
+
+ # Extract numpy arrays
+ face_0 = np.array(face_0[0]) # (68, 2)
+ face_ref = np.array(face_ref[0])
+
+ # Center point (nose tip or face center)
+ center_idx = 30
+ center_0 = face_0[center_idx]
+ center_ref = face_ref[center_idx]
+
+ # Calculate distance to center point
+ dist = np.linalg.norm(face_0 - center_0, axis=1)
+ dist_ref = np.linalg.norm(face_ref - center_ref, axis=1)
+
+ # Avoid the 0 distance of the center point itself
+ dist = np.delete(dist, center_idx)
+ dist_ref = np.delete(dist_ref, center_idx)
+
+ mean_dist = np.mean(dist)
+ mean_dist_ref = np.mean(dist_ref)
+
+ if mean_dist < 1e-6:
+ scale_n = 1.0
+ else:
+ scale_n = mean_dist_ref / mean_dist
+
+ # Clamp to [0.8, 1.5]
+ scale_n = np.clip(scale_n, 0.8, 1.5)
+
+ for i, pose in enumerate(poses):
+ face = pose['faces']
+ # Extract numpy array
+ face = np.array(face[0]) # (68, 2)
+ center = face[center_idx]
+ scaled_face = (face - center) * scale_n + center
+ poses[i]['faces'][0] = scaled_face
+
+ body = pose['bodies']
+ candidate = body['candidate']
+ candidate_np = np.array(candidate[0]) # (14, 2)
+ body_center = candidate_np[0]
+ scaled_candidate = (candidate_np - body_center) * scale_n + body_center
+ poses[i]['bodies']['candidate'][0] = scaled_candidate
+
+ # In-place modification
+ pose['faces'][0] = scaled_face
+
+ return scale_n
+
+class PoseDetectionVitPoseToDWPose:
+ @classmethod
+ def INPUT_TYPES(s):
+ return {
+ "required": {
+ "vitpose_model": ("POSEMODEL",),
+ "images": ("IMAGE",),
+ },
+ }
+
+ RETURN_TYPES = ("DWPOSES",)
+ RETURN_NAMES = ("dw_poses",)
+ FUNCTION = "process"
+ CATEGORY = "WanAnimatePreprocess"
+ DESCRIPTION = "ViTPose to DWPose format pose detection node."
+
+ def process(self, vitpose_model, images):
+
+ detector = vitpose_model["yolo"]
+ pose_model = vitpose_model["vitpose"]
+ B, H, W, C = images.shape
+
+ shape = np.array([H, W])[None]
+ images_np = images.numpy()
+
+ IMG_NORM_MEAN = np.array([0.485, 0.456, 0.406])
+ IMG_NORM_STD = np.array([0.229, 0.224, 0.225])
+ input_resolution=(256, 192)
+ rescale = 1.25
+
+ detector.reinit()
+ pose_model.reinit()
+
+ comfy_pbar = ProgressBar(B*2)
+ progress = 0
+ bboxes = []
+ for img in tqdm(images_np, total=len(images_np), desc="Detecting bboxes"):
+ bboxes.append(detector(
+ cv2.resize(img, (640, 640)).transpose(2, 0, 1)[None],
+ shape
+ )[0][0]["bbox"])
+ progress += 1
+ if progress % 10 == 0:
+ comfy_pbar.update_absolute(progress)
+
+ detector.cleanup()
+
+ kp2ds = []
+ for img, bbox in tqdm(zip(images_np, bboxes), total=len(images_np), desc="Extracting keypoints"):
+ if bbox is None or bbox[-1] <= 0 or (bbox[2] - bbox[0]) < 10 or (bbox[3] - bbox[1]) < 10:
+ bbox = np.array([0, 0, img.shape[1], img.shape[0]])
+
+ bbox_xywh = bbox
+ center, scale = bbox_from_detector(bbox_xywh, input_resolution, rescale=rescale)
+ img = crop(img, center, scale, (input_resolution[0], input_resolution[1]))[0]
+
+ img_norm = (img - IMG_NORM_MEAN) / IMG_NORM_STD
+ img_norm = img_norm.transpose(2, 0, 1).astype(np.float32)
+
+ keypoints = pose_model(img_norm[None], np.array(center)[None], np.array(scale)[None])
+ kp2ds.append(keypoints)
+ progress += 1
+ if progress % 10 == 0:
+ comfy_pbar.update_absolute(progress)
+
+ pose_model.cleanup()
+
+ kp2ds = np.concatenate(kp2ds, 0)
+ pose_metas = load_pose_metas_from_kp2ds_seq(kp2ds, width=W, height=H)
+ dwposes = [aaposemeta_to_dwpose_scail(meta) for meta in pose_metas]
+
+ return (dwposes,)
+
+
+class RenderNLFPoses:
+ @classmethod
+ def INPUT_TYPES(s):
+ return {"required": {
+ "nlf_poses": ("NLFPRED", {"tooltip": "Input poses for the model"}),
+ "width": ("INT", {"default": 512}),
+ "height": ("INT", {"default": 512}),
+ },
+ "optional": {
+ "dw_poses": ("DWPOSES", {"default": None, "tooltip": "Optional DW pose model for 2D drawing"}),
+ "ref_dw_pose": ("DWPOSES", {"default": None, "tooltip": "Optional reference DW pose model for alignment"}),
+ }
+ }
+
+ RETURN_TYPES = ("IMAGE", "MASK",)
+ RETURN_NAMES = ("image", "mask",)
+ FUNCTION = "predict"
+ CATEGORY = "WanVideoWrapper"
+
+ def predict(self, nlf_poses, width, height, dw_poses=None, ref_dw_pose=None):
+
+ from .NLFPoseExtract.nlf_render import render_nlf_as_images, shift_dwpose_according_to_nlf, process_data_to_COCO_format, intrinsic_matrix_from_field_of_view
+ from .NLFPoseExtract.align3d import solve_new_camera_params_central, solve_new_camera_params_down
+
+ if isinstance(nlf_poses, dict):
+ pose_input = nlf_poses['joints3d_nonparam'][0] if 'joints3d_nonparam' in nlf_poses else nlf_poses
+ else:
+ pose_input = nlf_poses
+
+ dw_pose_input = copy.deepcopy(dw_poses)
+
+ ori_camera_pose = intrinsic_matrix_from_field_of_view([height, width])
+ ori_focal = ori_camera_pose[0, 0]
+
+ if ref_dw_pose is not None:
+ ref_dw_pose_input = copy.deepcopy(ref_dw_pose)
+ pose_3d_first_driving_frame = pose_input[0][0].cpu().numpy()
+ pose_3d_coco_first_driving_frame = process_data_to_COCO_format(pose_3d_first_driving_frame)
+ poses_2d_ref = ref_dw_pose_input[0]['bodies']['candidate'][0][:14]
+ poses_2d_ref[:, 0] = poses_2d_ref[:, 0] * width
+ poses_2d_ref[:, 1] = poses_2d_ref[:, 1] * height
+
+ poses_2d_subset = ref_dw_pose[0]['bodies']['subset'][0][:14]
+ pose_3d_coco_first_driving_frame = pose_3d_coco_first_driving_frame[:14]
+
+ valid_indices, valid_upper_indices, valid_lower_indices = [], [], []
+ upper_body_indices = [0, 2, 3, 5, 6]
+ lower_body_indices = [9, 10, 12, 13]
+
+ for i in range(len(poses_2d_subset)):
+ if poses_2d_subset[i] != -1.0 and np.sum(pose_3d_coco_first_driving_frame[i]) != 0:
+ if i in upper_body_indices:
+ valid_upper_indices.append(i)
+ if i in lower_body_indices:
+ valid_lower_indices.append(i)
+
+ valid_indices = [1] + valid_lower_indices if len(valid_upper_indices) < 4 else [1] + valid_lower_indices + valid_upper_indices # align body or only lower body
+
+ pose_2d_ref = poses_2d_ref[valid_indices]
+ pose_3d_coco_first_driving_frame = pose_3d_coco_first_driving_frame[valid_indices]
+
+ if len(valid_lower_indices) >= 4:
+ new_camera_intrinsics, scale_m = solve_new_camera_params_down(pose_3d_coco_first_driving_frame, ori_focal, [height, width], pose_2d_ref)
+ else:
+ new_camera_intrinsics, scale_m = solve_new_camera_params_central(pose_3d_coco_first_driving_frame, ori_focal, [height, width], pose_2d_ref)
+
+ scale_face = scale_faces(list(dw_pose_input), list(ref_dw_pose)) # poses[0]['faces'].shape: 1, 68, 2 , poses_ref[0]['faces'].shape: 1, 68, 2
+
+ logging.info(f"Scale - m: {scale_m}, face: {scale_face}")
+ shift_dwpose_according_to_nlf(pose_input, dw_pose_input, ori_camera_pose, new_camera_intrinsics, height, width)
+
+ frames_np = render_nlf_as_images(pose_input, dw_pose_input, height, width, len(pose_input), intrinsic_matrix=new_camera_intrinsics)
+ else:
+ frames_np = render_nlf_as_images(pose_input, dw_pose_input, height, width, len(pose_input), intrinsic_matrix=ori_camera_pose)
+
+ frames_tensor = torch.from_numpy(np.stack(frames_np, axis=0)).contiguous() / 255.0
+ frames_tensor, mask = frames_tensor[..., :3], frames_tensor[..., -1] > 0.5
+
+ return (frames_tensor.cpu().float(), mask.cpu().float())
+
+NODE_CLASS_MAPPINGS = {
+ "PoseDetectionVitPoseToDWPose": PoseDetectionVitPoseToDWPose,
+ "RenderNLFPoses": RenderNLFPoses,
+}
+NODE_DISPLAY_NAME_MAPPINGS = {
+ "PoseDetectionVitPoseToDWPose": "Pose Detection VitPose to DWPose",
+ "RenderNLFPoses": "Render NLF Poses",
+}
diff --git a/pose_draw/draw_3d_utils.py b/pose_draw/draw_3d_utils.py
new file mode 100644
index 0000000..5af2725
--- /dev/null
+++ b/pose_draw/draw_3d_utils.py
@@ -0,0 +1,221 @@
+import numpy as np
+
+def convert_3dpose_to_2dpose_body(body_keypoints, face_keypoints):
+ """
+ Map 20-point 3D coordinates to 18-point 2D coordinates.
+ :param poses: Input list of 20 coordinates, each point as [x, y, z]
+ :return: Mapped list of 18 coordinates, each point as [x, y]
+ """
+ # Mapping relationship: index positions
+ body_mapping = {
+ 0: 1, 1: 2, 2: 3, 3: 4, 4: 5, 5: 6, 6: 7, 7: 8, 8: 8, 9: 9, 10: 10, 11: 23, 13: 22, 12: 21,
+ 14: 11, 15: 12, 16: 13, 17: 20, 18: 18, 19: 19
+ }
+ face_mapping = {
+ 1: 16, 8: 14, 4: 0, 7: 15, 0: 17
+ }
+
+ # Initialize 18-point coordinate list, default value is [-1, -1]
+ result = [[-1, -1] for _ in range(24)]
+
+ # Traverse the mapping relationship, map the corresponding 20-point coordinates to 18-point coordinates
+ for src_idx, dst_idx in body_mapping.items():
+ if src_idx < len(body_keypoints): # 确保索引不越界
+ result[dst_idx] = [body_keypoints[src_idx][1],body_keypoints[src_idx][0]] # Extract x, y coordinates
+ for src_idx, dst_idx in face_mapping.items():
+ if src_idx < len(face_keypoints):
+ result[dst_idx] = [face_keypoints[src_idx][1], face_keypoints[src_idx][0]]
+ return result
+
+def convert_3dpose_to_2dpose_hand(left_hand_keypoints, right_hand_keypoints, body_keypoints):
+ """
+ Map 20-point 3D coordinates to 18-point 2D coordinates.
+ :param poses: Input list of 20 coordinates, each point as [x, y, z]
+ :return: Mapped list of 18 coordinates, each point as [x, y]
+ """
+ # Mapping relationship: index positions
+ hand_mapping = {
+ 0: 1, 1: 2, 2: 3, 3: 4, 4: 5, 5: 6, 6: 7, 7: 8, 8: 9, 9: 10,
+ 10: 11, 11: 12, 12: 13, 13: 14, 14: 15, 15: 16, 16: 17, 17: 18,
+ 18: 19, 19: 20
+ }
+
+ body_mapping_left = {3: 0}
+ body_mapping_right = {6: 0}
+
+ # Initialize 18-point coordinate list, default value is [-1, -1]
+ left_result = [[-1, -1] for _ in range(21)]
+ right_result = [[-1, -1] for _ in range(21)]
+
+ # Traverse the mapping relationship, map the corresponding 20-point coordinates to 18-point coordinates
+ for src_idx, dst_idx in hand_mapping.items():
+ if src_idx < len(left_hand_keypoints): # 确保索引不越界
+ left_result[dst_idx] = [left_hand_keypoints[src_idx][1], left_hand_keypoints[src_idx][0]] # Extract x, y coordinates
+ right_result[dst_idx] = [right_hand_keypoints[src_idx][1], right_hand_keypoints[src_idx][0]]
+
+ for src_idx, dst_idx in body_mapping_left.items():
+ if src_idx < len(body_keypoints):
+ left_result[dst_idx] = [body_keypoints[src_idx][1], body_keypoints[src_idx][0]]
+ for src_idx, dst_idx in body_mapping_right.items():
+ if src_idx < len(body_keypoints):
+ right_result[dst_idx] = [body_keypoints[src_idx][1], body_keypoints[src_idx][0]]
+
+ return [left_result, right_result]
+
+def convert_3dpose_to_2dpose_face(face_keypoints):
+ # Set [-1, -1] for indices 0, 1, 4, 5, 6, 7, 8, otherwise extract [y, x]
+ result = [[-1, -1] if i in [0, 1, 4, 5, 6, 7, 8] else [pt[1], pt[0]] for i, pt in enumerate(face_keypoints)]
+ return result
+
+def correct_lift_end_kpt_by_phmr(start, end, dwpose_kpts, lift_start, lift_end, phmr_start, phmr_end):
+ '''
+ Check if the other end meets the requirements. If so, return the result after lift, otherwise return the phmr result.
+ '''
+ if dwpose_kpts[start][0] == -1:
+ return
+ lift_vec = np.array(lift_end) - np.array(lift_start)
+ phmr_vec = np.array(phmr_end) - np.array(phmr_start)
+ start_distance = np.linalg.norm(np.array(lift_start) - np.array(phmr_start))
+ end_distance = np.linalg.norm(np.array(lift_end) - np.array(phmr_end))
+ lift_vec_len = np.linalg.norm(lift_vec)
+ phmr_vec_len = np.linalg.norm(phmr_vec)
+ if start_distance + end_distance > phmr_vec_len:
+ dwpose_kpts[end] = [-1, -1]
+ theta = np.arccos(np.dot(lift_vec, phmr_vec) / (lift_vec_len * phmr_vec_len))
+ if lift_vec_len > phmr_vec_len * 1.65 or lift_vec_len < phmr_vec_len * 0.4 or theta > np.pi / 4:
+ dwpose_kpts[end] = [-1, -1]
+ return
+
+
+
+def mix_3d_poses(poses_dwpose, poses_3dpose):
+ '''
+ Combine two types of poses: use the body from 3dPose, and the face and hand from DWPose.
+ '''
+ poses = []
+ for pose_dwpose, pose_3dpose in zip(poses_dwpose, poses_3dpose):
+ pose = {
+ "bodies": {
+ "candidate": pose_3dpose["bodies"]["candidate"],
+ "subset": pose_dwpose["bodies"]["subset"]
+ },
+ "faces": pose_dwpose["faces"],
+ "hands": pose_dwpose["hands"]
+ }
+ poses.append(pose)
+ return poses
+
+def correct_hand_from_3d(hand_keypoints_dwpose, hand_keypoints_3dpose):
+ '''
+ If the hand keypoints of dwpose and 3dpose differ too much, remove the farthest end.
+ '''
+ edges_palm = [
+ [1, 2], [2, 3], [3, 4],
+ [5, 6], [6, 7], [7, 8],
+ [9, 10], [10, 11], [11, 12],
+ [13, 14], [14, 15], [15, 16],
+ [17, 18], [18, 19], [19, 20],
+ ]
+ edges_finger = [[0, 1], [0, 5], [0, 9], [0, 13], [0, 17]]
+ max_length_palm = 0
+ max_length_finger = 0
+ for edge in edges_palm:
+ limb_length_3dpose = np.linalg.norm(np.array(hand_keypoints_3dpose[edge[0]]) - np.array(hand_keypoints_3dpose[edge[1]]))
+ if limb_length_3dpose > max_length_palm:
+ max_length_palm = limb_length_3dpose
+ for edge in edges_finger:
+ limb_length_3dpose = np.linalg.norm(np.array(hand_keypoints_3dpose[edge[0]]) - np.array(hand_keypoints_3dpose[edge[1]]))
+ if limb_length_3dpose > max_length_finger:
+ max_length_finger = limb_length_3dpose
+ for edge in edges_palm:
+ limb_length_dwpose = np.linalg.norm(np.array(hand_keypoints_dwpose[edge[0]]) - np.array(hand_keypoints_dwpose[edge[1]]))
+ if limb_length_dwpose > max_length_palm * 1.5:
+ if -1 in hand_keypoints_dwpose[edge[0]] or -1 in hand_keypoints_dwpose[edge[1]] or -1 in hand_keypoints_3dpose[edge[0]] or -1 in hand_keypoints_3dpose[edge[1]]:
+ continue
+ distance_point_0 = np.linalg.norm(np.array(hand_keypoints_dwpose[edge[0]]) - np.array(hand_keypoints_3dpose[edge[0]]))
+ distance_point_1 = np.linalg.norm(np.array(hand_keypoints_dwpose[edge[1]]) - np.array(hand_keypoints_3dpose[edge[1]]))
+ if distance_point_0 > distance_point_1:
+ hand_keypoints_dwpose[edge[1]] = [-1, -1]
+ else:
+ hand_keypoints_dwpose[edge[0]] = [-1, -1]
+ for edge in edges_finger:
+ limb_length_dwpose = np.linalg.norm(np.array(hand_keypoints_dwpose[edge[0]]) - np.array(hand_keypoints_dwpose[edge[1]]))
+ if limb_length_dwpose > max_length_finger * 1.5:
+ if -1 in hand_keypoints_dwpose[edge[0]] or -1 in hand_keypoints_dwpose[edge[1]] or -1 in hand_keypoints_3dpose[edge[0]] or -1 in hand_keypoints_3dpose[edge[1]]:
+ continue
+ distance_point_0 = np.linalg.norm(np.array(hand_keypoints_dwpose[edge[0]]) - np.array(hand_keypoints_3dpose[edge[0]]))
+ distance_point_1 = np.linalg.norm(np.array(hand_keypoints_dwpose[edge[1]]) - np.array(hand_keypoints_3dpose[edge[1]]))
+ if distance_point_0 > distance_point_1:
+ hand_keypoints_dwpose[edge[1]] = [-1, -1]
+ else:
+ hand_keypoints_dwpose[edge[0]] = [-1, -1]
+ return hand_keypoints_dwpose
+
+def correct_body_from_3d(body_keypoints_dwpose, body_keypoints_3dpose, subset_dwpose, subset_3dpose):
+ '''
+ If the bone length of dwpose and 3dpose differ too much, remove the farthest end.
+ '''
+ limbSeq = [
+ [2, 3],
+ [2, 6],
+ [3, 4],
+ [4, 5],
+ [6, 7],
+ [7, 8],
+ [2, 9],
+ [9, 10],
+ [10, 11],
+ [2, 12],
+ [12, 13],
+ [13, 14],
+ [2, 1],
+ [1, 15],
+ [15, 17],
+ [1, 16],
+ [16, 18],
+ [3, 17],
+ [6, 18],
+ ]
+
+ for ori_limb in limbSeq:
+ limb = [ori_limb[0] - 1, ori_limb[1] - 1]
+ limb_length_dwpose = np.linalg.norm(np.array(body_keypoints_dwpose[limb[0]]) - np.array(body_keypoints_dwpose[limb[1]]))
+ limb_length_3dpose = np.linalg.norm(np.array(body_keypoints_3dpose[limb[0]]) - np.array(body_keypoints_3dpose[limb[1]]))
+ if subset_dwpose[0][limb[0]] == -1 or subset_dwpose[0][limb[1]] == -1 or subset_3dpose[0][limb[0]] == -1 or subset_3dpose[0][limb[1]] == -1:
+ continue
+ if limb_length_dwpose > limb_length_3dpose * 2:
+ # Determine the farther end
+ distance_point_0 = np.linalg.norm(np.array(body_keypoints_dwpose[limb[0]]) - np.array(body_keypoints_3dpose[limb[0]]))
+ distance_point_1 = np.linalg.norm(np.array(body_keypoints_dwpose[limb[1]]) - np.array(body_keypoints_3dpose[limb[1]]))
+ if distance_point_0 > distance_point_1:
+ if limb[1] == 1: # core
+ continue
+ body_keypoints_dwpose[limb[1]] = [-1, -1]
+ subset_dwpose[0][limb[1]] = -1
+ else:
+ if limb[0] == 1: # core
+ continue
+ body_keypoints_dwpose[limb[0]] = [-1, -1]
+ subset_dwpose[0][limb[0]] = -1
+ return body_keypoints_dwpose, subset_dwpose
+
+def correct_full_pose_from_3d(poses_dwpose, poses_3dpose):
+ '''
+ If the bone length of dwpose and 3dpose differ too much, remove the end farthest from the 3d pose.
+ '''
+ poses = []
+ for pose_dwpose, pose_3dpose in zip(poses_dwpose, poses_3dpose):
+ new_candidate, new_subset = correct_body_from_3d(pose_dwpose["bodies"]["candidate"], pose_3dpose["bodies"]["candidate"], pose_dwpose["bodies"]["subset"], pose_3dpose["bodies"]["subset"])
+ new_hands_0 = correct_hand_from_3d(pose_dwpose["hands"][0], pose_3dpose["hands"][0])
+ new_hands_1 = correct_hand_from_3d(pose_dwpose["hands"][1], pose_3dpose["hands"][1])
+ pose = {
+ "bodies": {
+ "candidate": new_candidate,
+ "subset": new_subset
+ },
+ "faces": pose_dwpose["faces"],
+ "hands": [new_hands_0, new_hands_1]
+ }
+ poses.append(pose)
+
+ return poses
diff --git a/pose_draw/draw_pose_utils.py b/pose_draw/draw_pose_utils.py
new file mode 100644
index 0000000..16b4592
--- /dev/null
+++ b/pose_draw/draw_pose_utils.py
@@ -0,0 +1,129 @@
+import cv2
+import numpy as np
+from PIL import Image
+import os
+from .draw_utils import draw_bodypose, draw_bodypose_with_feet, draw_handpose_lr, draw_handpose, draw_facepose, draw_bodypose_augmentation
+
+
+def draw_pose(pose, H, W, show_feet=False, show_body=True, show_hand=True, show_face=True, show_cheek=False, dw_bgr=False, dw_hand=False, aug_body_draw=False, optimized_face=False):
+ final_canvas = np.zeros(shape=(H, W, 3), dtype=np.uint8)
+ for i in range(len(pose["bodies"]["candidate"])):
+ canvas = np.zeros(shape=(H, W, 3), dtype=np.uint8)
+ bodies = pose["bodies"]
+ faces = pose["faces"][i:i+1]
+ hands = pose["hands"][2*i:2*i+2]
+ candidate = bodies["candidate"][i]
+ subset = bodies["subset"][i:i+1]
+
+ if show_body:
+ if len(subset[0]) <= 18 or show_feet == False:
+ if aug_body_draw:
+ raise NotImplementedError("aug_body_draw is not implemented yet")
+ else:
+ canvas = draw_bodypose(canvas, candidate, subset)
+ else:
+ canvas = draw_bodypose_with_feet(canvas, candidate, subset)
+ if dw_bgr:
+ canvas = cv2.cvtColor(canvas, cv2.COLOR_BGR2RGB)
+ if show_cheek:
+ assert show_body == False, "show_cheek and show_body cannot be True at the same time"
+ canvas = draw_bodypose_augmentation(canvas, candidate, subset, drop_aug=True, shift_aug=False, all_cheek_aug=True)
+ if show_hand:
+ if not dw_hand:
+ canvas = draw_handpose_lr(canvas, hands)
+ else:
+ canvas = draw_handpose(canvas, hands)
+ if show_face:
+ canvas = draw_facepose(canvas, faces, optimized_face=optimized_face)
+ final_canvas = final_canvas + canvas
+ return final_canvas
+
+
+def scale_image_hw_keep_size(img, scale_h, scale_w):
+ """Scale the image by scale_h and scale_w respectively, keeping the output size unchanged."""
+ H, W = img.shape[:2]
+ new_H, new_W = int(H * scale_h), int(W * scale_w)
+ scaled = cv2.resize(img, (new_W, new_H), interpolation=cv2.INTER_LINEAR)
+
+ result = np.zeros_like(img)
+
+ # 计算在目标图上的放置范围
+ # --- Y方向 ---
+ if new_H >= H:
+ y_start_src = (new_H - H) // 2
+ y_end_src = y_start_src + H
+ y_start_dst = 0
+ y_end_dst = H
+ else:
+ y_start_src = 0
+ y_end_src = new_H
+ y_start_dst = (H - new_H) // 2
+ y_end_dst = y_start_dst + new_H
+
+ # --- X方向 ---
+ if new_W >= W:
+ x_start_src = (new_W - W) // 2
+ x_end_src = x_start_src + W
+ x_start_dst = 0
+ x_end_dst = W
+ else:
+ x_start_src = 0
+ x_end_src = new_W
+ x_start_dst = (W - new_W) // 2
+ x_end_dst = x_start_dst + new_W
+
+ # 将 scaled 映射到 result
+ result[y_start_dst:y_end_dst, x_start_dst:x_end_dst] = scaled[y_start_src:y_end_src, x_start_src:x_end_src]
+
+ return result
+
+def draw_pose_to_canvas_np(poses, pool, H, W, reshape_scale, show_feet_flag=False, show_body_flag=True, show_hand_flag=True, show_face_flag=True, show_cheek_flag=False, dw_bgr=False, dw_hand=False, aug_body_draw=False):
+ canvas_np_lst = []
+ for pose in poses:
+ if reshape_scale > 0:
+ pool.apply_random_reshapes(pose)
+ canvas = draw_pose(pose, H, W, show_feet_flag, show_body_flag, show_hand_flag, show_face_flag, show_cheek_flag, dw_bgr, dw_hand, aug_body_draw, optimized_face=True)
+ canvas_np_lst.append(canvas)
+ return canvas_np_lst
+
+
+def draw_pose_to_canvas(poses, pool, H, W, reshape_scale, points_only_flag, show_feet_flag, show_body_flag=True, show_hand_flag=True, show_face_flag=True, show_cheek_flag=False, dw_bgr=False, dw_hand=False, aug_body_draw=False):
+ canvas_lst = []
+ for pose in poses:
+ if reshape_scale > 0:
+ pool.apply_random_reshapes(pose)
+ canvas = draw_pose(pose, H, W, show_feet_flag, show_body_flag, show_hand_flag, show_face_flag, show_cheek_flag, dw_bgr, dw_hand, aug_body_draw, optimized_face=False)
+ canvas_img = Image.fromarray(canvas)
+ canvas_lst.append(canvas_img)
+ return canvas_lst
+
+
+def get_mp4_filenames_from_directory(dwpose_keypoints_dir):
+ mp4_filenames_dwpose = []
+ # Get all available mp4 files by intersecting keypoints and mp4
+ if dwpose_keypoints_dir:
+ for root, dirs, files in os.walk(dwpose_keypoints_dir):
+ for file in files:
+ if file.lower().endswith('.pt'): # Only look for .mp4 files
+ mp4_filenames_dwpose.append(file.replace(".pt", ".mp4")) # Get absolute path
+ return mp4_filenames_dwpose
+
+def project_dwpose_to_3d(dwpose_keypoint, original_threed_keypoint, focal, princpt, H, W):
+ # Camera intrinsic parameters
+ # fx, fy = focal, focal
+ fx, fy = focal
+ cx, cy = princpt
+
+ # 2D keypoint coordinates
+ x_2d, y_2d = dwpose_keypoint[0] * W, dwpose_keypoint[1] * H
+
+ # Original 3D point (in camera coordinate system)
+ ori_x, ori_y, ori_z = original_threed_keypoint
+
+ # Use the new 2D point and original depth to compute the new 3D point by back-projection
+ # Formula: x = (u - cx) * z / fx
+ new_x = (x_2d - cx) * ori_z / fx
+ new_y = (y_2d - cy) * ori_z / fy
+ new_z = ori_z # Keep the depth unchanged
+
+ return [new_x, new_y, new_z]
diff --git a/pose_draw/draw_utils.py b/pose_draw/draw_utils.py
new file mode 100644
index 0000000..84e6e4f
--- /dev/null
+++ b/pose_draw/draw_utils.py
@@ -0,0 +1,658 @@
+# https://github.com/IDEA-Research/DWPose
+import math
+import numpy as np
+import matplotlib
+import cv2
+import random
+
+eps = 0.01
+
+
+def smart_resize(x, s):
+ Ht, Wt = s
+ if x.ndim == 2:
+ Ho, Wo = x.shape
+ Co = 1
+ else:
+ Ho, Wo, Co = x.shape
+ if Co == 3 or Co == 1:
+ k = float(Ht + Wt) / float(Ho + Wo)
+ return cv2.resize(
+ x,
+ (int(Wt), int(Ht)),
+ interpolation=cv2.INTER_AREA if k < 1 else cv2.INTER_LANCZOS4,
+ )
+ else:
+ return np.stack([smart_resize(x[:, :, i], s) for i in range(Co)], axis=2)
+
+
+def smart_resize_k(x, fx, fy):
+ if x.ndim == 2:
+ Ho, Wo = x.shape
+ Co = 1
+ else:
+ Ho, Wo, Co = x.shape
+ Ht, Wt = Ho * fy, Wo * fx
+ if Co == 3 or Co == 1:
+ k = float(Ht + Wt) / float(Ho + Wo)
+ return cv2.resize(
+ x,
+ (int(Wt), int(Ht)),
+ interpolation=cv2.INTER_AREA if k < 1 else cv2.INTER_LANCZOS4,
+ )
+ else:
+ return np.stack([smart_resize_k(x[:, :, i], fx, fy) for i in range(Co)], axis=2)
+
+
+def padRightDownCorner(img, stride, padValue):
+ h = img.shape[0]
+ w = img.shape[1]
+
+ pad = 4 * [None]
+ pad[0] = 0 # up
+ pad[1] = 0 # left
+ pad[2] = 0 if (h % stride == 0) else stride - (h % stride) # down
+ pad[3] = 0 if (w % stride == 0) else stride - (w % stride) # right
+
+ img_padded = img
+ pad_up = np.tile(img_padded[0:1, :, :] * 0 + padValue, (pad[0], 1, 1))
+ img_padded = np.concatenate((pad_up, img_padded), axis=0)
+ pad_left = np.tile(img_padded[:, 0:1, :] * 0 + padValue, (1, pad[1], 1))
+ img_padded = np.concatenate((pad_left, img_padded), axis=1)
+ pad_down = np.tile(img_padded[-2:-1, :, :] * 0 + padValue, (pad[2], 1, 1))
+ img_padded = np.concatenate((img_padded, pad_down), axis=0)
+ pad_right = np.tile(img_padded[:, -2:-1, :] * 0 + padValue, (1, pad[3], 1))
+ img_padded = np.concatenate((img_padded, pad_right), axis=1)
+
+ return img_padded, pad
+
+
+def transfer(model, model_weights):
+ transfered_model_weights = {}
+ for weights_name in model.state_dict().keys():
+ transfered_model_weights[weights_name] = model_weights[
+ ".".join(weights_name.split(".")[1:])
+ ]
+ return transfered_model_weights
+
+def draw_bodypose_with_feet(canvas, candidate, subset):
+ H, W, C = canvas.shape
+ candidate = np.array(candidate)
+ subset = np.array(subset)
+
+ stickwidth = 4
+
+ # 原始18个关节点的连接顺序(和 OpenPose 的 COCO 模型一致)
+ limbSeq = [
+ [2, 3],
+ [2, 6],
+ [3, 4],
+ [4, 5],
+ [6, 7],
+ [7, 8],
+ [2, 9],
+ [9, 10],
+ [10, 11],
+ [2, 12],
+ [12, 13],
+ [13, 14],
+ [2, 1],
+ [1, 15],
+ [15, 17],
+ [1, 16],
+ [16, 18],
+ [3, 17],
+ [6, 18],
+ ]
+
+ # 添加脚部连接线:10->18, 10->19, 10->20;13->21, 13->22, 13->23
+ foot_limbSeq = [
+ [14, 19],
+ [14, 20],
+ [14, 21],
+ [11, 22],
+ [11, 23],
+ [11, 24],
+ ]
+
+ # 生成颜色(原始18条颜色 + 6条新颜色)
+ colors = [
+ [255, 0, 0],
+ [255, 85, 0],
+ [255, 170, 0],
+ [255, 255, 0],
+ [170, 255, 0],
+ [85, 255, 0],
+ [0, 255, 0],
+ [0, 255, 85],
+ [0, 255, 170],
+ [0, 255, 255],
+ [0, 170, 255],
+ [0, 85, 255],
+ [0, 0, 255],
+ [85, 0, 255],
+ [170, 0, 255],
+ [255, 0, 255],
+ [255, 0, 170],
+ [255, 0, 85],
+ ]
+
+ colors_feet = [
+ [100, 0, 215], [80, 0, 235], [60, 0, 255],
+ [0, 235, 150], [0, 215, 170], [0, 195, 190],
+ ]
+
+ colors = colors + colors_feet
+
+ for i in range(17):
+ for n in range(len(subset)):
+ index = subset[n][np.array(limbSeq[i]) - 1]
+ if -1 in index:
+ continue
+ Y = candidate[index.astype(int), 0] * float(W)
+ X = candidate[index.astype(int), 1] * float(H)
+ mX = np.mean(X)
+ mY = np.mean(Y)
+ length = ((X[0] - X[1]) ** 2 + (Y[0] - Y[1]) ** 2) ** 0.5
+ angle = math.degrees(math.atan2(X[0] - X[1], Y[0] - Y[1]))
+ polygon = cv2.ellipse2Poly(
+ (int(mY), int(mX)), (int(length / 2), stickwidth), int(angle), 0, 360, 1
+ )
+ cv2.fillConvexPoly(canvas, polygon, colors[i])
+
+ for i in range(6):
+ for n in range(len(subset)):
+ index = subset[n][np.array(foot_limbSeq[i]) - 1]
+ if -1 in index:
+ continue
+ Y = candidate[index.astype(int), 0] * float(W)
+ X = candidate[index.astype(int), 1] * float(H)
+ mX = np.mean(X)
+ mY = np.mean(Y)
+ length = ((X[0] - X[1]) ** 2 + (Y[0] - Y[1]) ** 2) ** 0.5
+ angle = math.degrees(math.atan2(X[0] - X[1], Y[0] - Y[1]))
+ polygon = cv2.ellipse2Poly(
+ (int(mY), int(mX)), (int(length / 2), stickwidth), int(angle), 0, 360, 1
+ )
+ cv2.fillConvexPoly(canvas, polygon, colors_feet[i])
+
+
+ canvas = (canvas * 0.6).astype(np.uint8)
+
+ # 画关键点
+ for i in range(24):
+ for n in range(len(subset)):
+ index = int(subset[n][i])
+ if index == -1:
+ continue
+ x, y = candidate[index][0:2]
+ x = int(x * W)
+ y = int(y * H)
+ cv2.circle(canvas, (int(x), int(y)), 4, colors[i], thickness=-1)
+ return canvas
+
+
+def draw_bodypose_augmentation(canvas, candidate, subset, drop_aug=True, shift_aug=False, all_cheek_aug=False):
+ H, W, C = canvas.shape
+ candidate = np.array(candidate)
+ subset = np.array(subset)
+
+ stickwidth = 4
+
+ limbSeq = [
+ [2, 3], # 1->2 左肩 0
+ [2, 6], # 1->5 右肩 1
+ [3, 4], # 2->3 左臂 2
+ [4, 5], # 3->4 左肘 3
+ [6, 7], # 5->6 右臂 4
+ [7, 8], # 6->7 右肘 5
+ [2, 9], # 6
+ [9, 10], # 7
+ [10, 11], # 8
+ [2, 12], # 9
+ [12, 13], # 10
+ [13, 14], # 11
+ [2, 1], # 12
+ [1, 15], # 13 cheek
+ [15, 17], # 14 cheek
+ [1, 16], # 15 cheek
+ [16, 18], # 16 cheek
+ [3, 17],
+ [6, 18],
+ ]
+
+ colors = [
+ [255, 0, 0],
+ [255, 85, 0],
+ [255, 170, 0],
+ [255, 255, 0],
+ [170, 255, 0],
+ [85, 255, 0],
+ [0, 255, 0],
+ [0, 255, 85],
+ [0, 255, 170],
+ [0, 255, 255],
+ [0, 170, 255],
+ [0, 85, 255],
+ [0, 0, 255],
+ [85, 0, 255],
+ [170, 0, 255],
+ [255, 0, 255],
+ [255, 0, 170],
+ [255, 0, 85],
+ ]
+
+ # 随机选0-2根骨骼进行丢弃
+ if drop_aug:
+ arr_drop = list(range(17))
+ k_drop = random.choices([0, 1, 2], weights=[0.5, 0.3, 0.2])[0]
+ drop_indices = random.sample(arr_drop, k_drop)
+ else:
+ drop_indices = []
+ if shift_aug:
+ shift_indices = random.sample(list(range(17)), 2)
+ else:
+ shift_indices = []
+ if all_cheek_aug:
+ drop_indices = list(range(13)) # 0-12对应的骨骼都扔掉
+
+ for i in range(17):
+ for n in range(len(subset)):
+ index = subset[n][np.array(limbSeq[i]) - 1]
+ if -1 in index:
+ continue
+ Y = candidate[index.astype(int), 0] * float(W)
+ X = candidate[index.astype(int), 1] * float(H)
+
+ if i in drop_indices:
+ continue
+
+ mX = np.mean(X) # 计算两个关节点之间的中点
+ mY = np.mean(Y)
+ length = ((X[0] - X[1]) ** 2 + (Y[0] - Y[1]) ** 2) ** 0.5
+ if i in shift_indices:
+ mX = mX + random.uniform(-length/4, length/4)
+ mY = mY + random.uniform(-length/4, length/4)
+ angle = math.degrees(math.atan2(X[0] - X[1], Y[0] - Y[1]))
+ polygon = cv2.ellipse2Poly(
+ (int(mY), int(mX)), (int(length / 2), stickwidth), int(angle), 0, 360, 1
+ )
+ cv2.fillConvexPoly(canvas, polygon, colors[i])
+
+ canvas = (canvas * 0.6).astype(np.uint8)
+
+ for i in range(18):
+ if all_cheek_aug:
+ if not i in [0, 14, 15, 16, 17]:
+ continue
+ for n in range(len(subset)):
+ index = int(subset[n][i])
+ if index == -1:
+ continue
+ x, y = candidate[index][0:2]
+ x = int(x * W)
+ y = int(y * H)
+ cv2.circle(canvas, (int(x), int(y)), 4, colors[i], thickness=-1)
+
+ return canvas
+
+def draw_bodypose(canvas, candidate, subset):
+ H, W, C = canvas.shape
+ candidate = np.array(candidate)
+ subset = np.array(subset)
+
+ stickwidth = 4
+
+ limbSeq = [
+ [2, 3],
+ [2, 6],
+ [3, 4],
+ [4, 5],
+ [6, 7],
+ [7, 8],
+ [2, 9],
+ [9, 10],
+ [10, 11],
+ [2, 12],
+ [12, 13],
+ [13, 14],
+ [2, 1],
+ [1, 15],
+ [15, 17],
+ [1, 16],
+ [16, 18],
+ [3, 17],
+ [6, 18],
+ ]
+
+ colors = [
+ [255, 0, 0],
+ [255, 85, 0],
+ [255, 170, 0],
+ [255, 255, 0],
+ [170, 255, 0],
+ [85, 255, 0],
+ [0, 255, 0],
+ [0, 255, 85],
+ [0, 255, 170],
+ [0, 255, 255],
+ [0, 170, 255],
+ [0, 85, 255],
+ [0, 0, 255],
+ [85, 0, 255],
+ [170, 0, 255],
+ [255, 0, 255],
+ [255, 0, 170],
+ [255, 0, 85],
+ ]
+
+ for i in range(17):
+ for n in range(len(subset)):
+ index = subset[n][np.array(limbSeq[i]) - 1]
+ if -1 in index:
+ continue
+ Y = candidate[index.astype(int), 0] * float(W)
+ X = candidate[index.astype(int), 1] * float(H)
+ mX = np.mean(X)
+ mY = np.mean(Y)
+ length = ((X[0] - X[1]) ** 2 + (Y[0] - Y[1]) ** 2) ** 0.5
+ angle = math.degrees(math.atan2(X[0] - X[1], Y[0] - Y[1]))
+ polygon = cv2.ellipse2Poly(
+ (int(mY), int(mX)), (int(length / 2), stickwidth), int(angle), 0, 360, 1
+ )
+ cv2.fillConvexPoly(canvas, polygon, colors[i])
+
+ canvas = (canvas * 0.6).astype(np.uint8)
+
+ for i in range(18):
+ for n in range(len(subset)):
+ index = int(subset[n][i])
+ if index == -1:
+ continue
+ x, y = candidate[index][0:2]
+ x = int(x * W)
+ y = int(y * H)
+ cv2.circle(canvas, (int(x), int(y)), 4, colors[i], thickness=-1)
+
+ return canvas
+
+def draw_handpose_lr(canvas, all_hand_peaks):
+ H, W, C = canvas.shape
+
+ # 连接顺序:21个关键点的骨架连线
+ edges = [
+ [0, 1], [1, 2], [2, 3], [3, 4],
+ [0, 5], [5, 6], [6, 7], [7, 8],
+ [0, 9], [9, 10], [10, 11], [11, 12],
+ [0, 13], [13, 14], [14, 15], [15, 16],
+ [0, 17], [17, 18], [18, 19], [19, 20],
+ ]
+
+ all_num_hands = len(all_hand_peaks)
+ for peaks_idx, peaks in enumerate(all_hand_peaks):
+ left_or_right = not (peaks_idx >= all_num_hands / 2)
+ base_hue = 0 if left_or_right == 0 else 0.3
+ peaks = np.array(peaks)
+
+ for ie, e in enumerate(edges):
+ x1, y1 = peaks[e[0]]
+ x2, y2 = peaks[e[1]]
+ x1 = int(x1 * W)
+ y1 = int(y1 * H)
+ x2 = int(x2 * W)
+ y2 = int(y2 * H)
+ if x1 > eps and y1 > eps and x2 > eps and y2 > eps:
+ if left_or_right == 0:
+ hsv_color = [ (base_hue + ie / float(len(edges)) * 0.8), 0.9, 0.9 ]
+ else:
+ hsv_color = [ (base_hue + ie / float(len(edges)) * 0.8), 0.8, 1 ]
+ rgb_color = matplotlib.colors.hsv_to_rgb(hsv_color) * 255
+ cv2.line(
+ canvas,
+ (x1, y1),
+ (x2, y2),
+ rgb_color,
+ thickness=2,
+ )
+
+ for i, keypoint in enumerate(peaks):
+ x, y = keypoint
+ x = int(x * W)
+ y = int(y * H)
+ if x > eps and y > eps:
+ # 关键点也用淡色标注(左手蓝、右手红)
+ point_color = (245, 100, 100) if left_or_right == 0 else (100, 100, 255)
+ cv2.circle(canvas, (x, y), 4, point_color, thickness=-1)
+
+ return canvas
+
+def draw_handpose(canvas, all_hand_peaks):
+ H, W, C = canvas.shape
+ stickwidth_thin = min(max(int(min(H, W) / 300), 1), 2)
+
+ edges = [
+ [0, 1],
+ [1, 2],
+ [2, 3],
+ [3, 4],
+ [0, 5],
+ [5, 6],
+ [6, 7],
+ [7, 8],
+ [0, 9],
+ [9, 10],
+ [10, 11],
+ [11, 12],
+ [0, 13],
+ [13, 14],
+ [14, 15],
+ [15, 16],
+ [0, 17],
+ [17, 18],
+ [18, 19],
+ [19, 20],
+ ]
+
+ for peaks in all_hand_peaks:
+ peaks = np.array(peaks)
+
+ for ie, e in enumerate(edges):
+ x1, y1 = peaks[e[0]]
+ x2, y2 = peaks[e[1]]
+ x1 = int(x1 * W)
+ y1 = int(y1 * H)
+ x2 = int(x2 * W)
+ y2 = int(y2 * H)
+ if x1 > eps and y1 > eps and x2 > eps and y2 > eps:
+ cv2.line(
+ canvas,
+ (x1, y1),
+ (x2, y2),
+ matplotlib.colors.hsv_to_rgb([ie / float(len(edges)), 1.0, 1.0])
+ * 255,
+ thickness=stickwidth_thin,
+ )
+
+ for i, keyponit in enumerate(peaks):
+ x, y = keyponit
+ x = int(x * W)
+ y = int(y * H)
+ if x > eps and y > eps:
+ cv2.circle(canvas, (x, y), stickwidth_thin, (0, 0, 255), thickness=-1)
+ return canvas
+
+
+def draw_facepose(canvas, all_lmks, optimized_face=True):
+ H, W, C = canvas.shape
+ stickwidth = min(max(int(min(H, W) / 200), 1), 3)
+ stickwidth_thin = min(max(int(min(H, W) / 300), 1), 2)
+
+ for lmks in all_lmks:
+ lmks = np.array(lmks)
+ for lmk_idx, lmk in enumerate(lmks):
+ x, y = lmk
+ x = int(x * W)
+ y = int(y * H)
+ if x > eps and y > eps:
+ if optimized_face:
+ if lmk_idx in list(range(17, 27)) + list(range(36, 70)):
+ cv2.circle(canvas, (x, y), stickwidth_thin, (255, 255, 255), thickness=-1)
+ else:
+ cv2.circle(canvas, (x, y), stickwidth, (255, 255, 255), thickness=-1)
+ return canvas
+
+
+
+
+
+# detect hand according to body pose keypoints
+# please refer to https://github.com/CMU-Perceptual-Computing-Lab/openpose/blob/master/src/openpose/hand/handDetector.cpp
+def handDetect(candidate, subset, oriImg):
+ # right hand: wrist 4, elbow 3, shoulder 2
+ # left hand: wrist 7, elbow 6, shoulder 5
+ ratioWristElbow = 0.33
+ detect_result = []
+ image_height, image_width = oriImg.shape[0:2]
+ for person in subset.astype(int):
+ # if any of three not detected
+ has_left = np.sum(person[[5, 6, 7]] == -1) == 0
+ has_right = np.sum(person[[2, 3, 4]] == -1) == 0
+ if not (has_left or has_right):
+ continue
+ hands = []
+ # left hand
+ if has_left:
+ left_shoulder_index, left_elbow_index, left_wrist_index = person[[5, 6, 7]]
+ x1, y1 = candidate[left_shoulder_index][:2]
+ x2, y2 = candidate[left_elbow_index][:2]
+ x3, y3 = candidate[left_wrist_index][:2]
+ hands.append([x1, y1, x2, y2, x3, y3, True])
+ # right hand
+ if has_right:
+ right_shoulder_index, right_elbow_index, right_wrist_index = person[
+ [2, 3, 4]
+ ]
+ x1, y1 = candidate[right_shoulder_index][:2]
+ x2, y2 = candidate[right_elbow_index][:2]
+ x3, y3 = candidate[right_wrist_index][:2]
+ hands.append([x1, y1, x2, y2, x3, y3, False])
+
+ for x1, y1, x2, y2, x3, y3, is_left in hands:
+ # pos_hand = pos_wrist + ratio * (pos_wrist - pos_elbox) = (1 + ratio) * pos_wrist - ratio * pos_elbox
+ # handRectangle.x = posePtr[wrist*3] + ratioWristElbow * (posePtr[wrist*3] - posePtr[elbow*3]);
+ # handRectangle.y = posePtr[wrist*3+1] + ratioWristElbow * (posePtr[wrist*3+1] - posePtr[elbow*3+1]);
+ # const auto distanceWristElbow = getDistance(poseKeypoints, person, wrist, elbow);
+ # const auto distanceElbowShoulder = getDistance(poseKeypoints, person, elbow, shoulder);
+ # handRectangle.width = 1.5f * fastMax(distanceWristElbow, 0.9f * distanceElbowShoulder);
+ x = x3 + ratioWristElbow * (x3 - x2)
+ y = y3 + ratioWristElbow * (y3 - y2)
+ distanceWristElbow = math.sqrt((x3 - x2) ** 2 + (y3 - y2) ** 2)
+ distanceElbowShoulder = math.sqrt((x2 - x1) ** 2 + (y2 - y1) ** 2)
+ width = 1.5 * max(distanceWristElbow, 0.9 * distanceElbowShoulder)
+ # x-y refers to the center --> offset to topLeft point
+ # handRectangle.x -= handRectangle.width / 2.f;
+ # handRectangle.y -= handRectangle.height / 2.f;
+ x -= width / 2
+ y -= width / 2 # width = height
+ # overflow the image
+ if x < 0:
+ x = 0
+ if y < 0:
+ y = 0
+ width1 = width
+ width2 = width
+ if x + width > image_width:
+ width1 = image_width - x
+ if y + width > image_height:
+ width2 = image_height - y
+ width = min(width1, width2)
+ # the max hand box value is 20 pixels
+ if width >= 20:
+ detect_result.append([int(x), int(y), int(width), is_left])
+
+ """
+ return value: [[x, y, w, True if left hand else False]].
+ width=height since the network require squared input.
+ x, y is the coordinate of top left
+ """
+ return detect_result
+
+
+# Written by Lvmin
+def faceDetect(candidate, subset, oriImg):
+ # left right eye ear 14 15 16 17
+ detect_result = []
+ image_height, image_width = oriImg.shape[0:2]
+ for person in subset.astype(int):
+ has_head = person[0] > -1
+ if not has_head:
+ continue
+
+ has_left_eye = person[14] > -1
+ has_right_eye = person[15] > -1
+ has_left_ear = person[16] > -1
+ has_right_ear = person[17] > -1
+
+ if not (has_left_eye or has_right_eye or has_left_ear or has_right_ear):
+ continue
+
+ head, left_eye, right_eye, left_ear, right_ear = person[[0, 14, 15, 16, 17]]
+
+ width = 0.0
+ x0, y0 = candidate[head][:2]
+
+ if has_left_eye:
+ x1, y1 = candidate[left_eye][:2]
+ d = max(abs(x0 - x1), abs(y0 - y1))
+ width = max(width, d * 3.0)
+
+ if has_right_eye:
+ x1, y1 = candidate[right_eye][:2]
+ d = max(abs(x0 - x1), abs(y0 - y1))
+ width = max(width, d * 3.0)
+
+ if has_left_ear:
+ x1, y1 = candidate[left_ear][:2]
+ d = max(abs(x0 - x1), abs(y0 - y1))
+ width = max(width, d * 1.5)
+
+ if has_right_ear:
+ x1, y1 = candidate[right_ear][:2]
+ d = max(abs(x0 - x1), abs(y0 - y1))
+ width = max(width, d * 1.5)
+
+ x, y = x0, y0
+
+ x -= width
+ y -= width
+
+ if x < 0:
+ x = 0
+
+ if y < 0:
+ y = 0
+
+ width1 = width * 2
+ width2 = width * 2
+
+ if x + width > image_width:
+ width1 = image_width - x
+
+ if y + width > image_height:
+ width2 = image_height - y
+
+ width = min(width1, width2)
+
+ if width >= 20:
+ detect_result.append([int(x), int(y), int(width)])
+
+ return detect_result
+
+
+# get max index of 2d array
+def npmax(array):
+ arrayindex = array.argmax(1)
+ arrayvalue = array.max(1)
+ i = arrayvalue.argmax()
+ j = arrayindex[i]
+ return i, j
diff --git a/pyproject.toml b/pyproject.toml
new file mode 100644
index 0000000..1bee94c
--- /dev/null
+++ b/pyproject.toml
@@ -0,0 +1,15 @@
+[project]
+name = "ComfyUI-SCAIL-Pose"
+description = "ComfyUI nodes for SCAIL input processing"
+version = "1.0.0"
+license = {file = "LICENSE"}
+dependencies = ["taichi", "pyrender", "trimesh", "opencv-python", "matplotlib", "pillow"]
+
+[project.urls]
+Repository = "https://github.com/kijai/ComfyUI-SCAIL-Pose"
+# Used by Comfy Registry https://comfyregistry.org
+
+[tool.comfy]
+PublisherId = "kijai"
+DisplayName = "ComfyUI-SCAIL-Pose"
+Icon = ""
diff --git a/readme.md b/readme.md
new file mode 100644
index 0000000..d5cf92e
--- /dev/null
+++ b/readme.md
@@ -0,0 +1,12 @@
+# ComfyUI nodes for SCAIL-pose processing
+
+
+The code is cleaned, simplified version of: https://github.com/zai-org/SCAIL-Pose
+
+For face and hands, instead of DWPose this uses Vitpose and it's outputs converted into DWpose format for the optional alignment
+
+VitPose detector is available in these nodes: https://github.com/kijai/ComfyUI-WanAnimatePreprocess
+
+NLF model loader is already included in WanVideoWrapper
+
+Reason this is separate repository is the additional requirements of `taichi` and `pyrender`
diff --git a/render_3d/render_cylinder.py b/render_3d/render_cylinder.py
new file mode 100644
index 0000000..65a9964
--- /dev/null
+++ b/render_3d/render_cylinder.py
@@ -0,0 +1,93 @@
+import numpy as np
+import cv2
+from PIL import Image
+import pyrender
+import trimesh
+
+def render_colored_cylinders(cylinder_specs, focal, princpt, image_size=(1280, 1280), img=None):
+
+ H, W = image_size
+ if isinstance(focal, float) or isinstance(focal, int):
+ fx, fy = focal, focal
+ else:
+ fx, fy = focal[0], focal[1]
+ cx, cy = princpt
+
+
+ # Initialize scene
+ scene = pyrender.Scene(bg_color=[0, 0, 0, 0], ambient_light=[0.1, 0.1, 0.1])
+
+
+ # Set up camera
+ camera = pyrender.IntrinsicsCamera(fx=fx, fy=fy, cx=cx, cy=cy, znear=0.5, zfar=10000)
+ pyrender2opencv = np.array([[1.0, 0, 0, 0],
+ [0, -1, 0, 0],
+ [0, 0, -1, 0],
+ [0, 0, 0, 1]])
+ cam_pose = pyrender2opencv @ np.eye(4)
+ scene.add(camera, pose=cam_pose)
+
+ # Add light source
+ light = pyrender.DirectionalLight(color=np.ones(3), intensity=3.0)
+ scene.add(light, pose=cam_pose)
+
+ points_to_draw = []
+
+ for start, end, color in cylinder_specs:
+ start = np.array(start)
+ end = np.array(end)
+ vec = end - start
+ height = np.linalg.norm(vec)
+ if height == 0:
+ continue
+
+ tm = trimesh.creation.cylinder(radius=12, height=height, sections=16)
+
+ # Rotate to align with z-axis
+ z_axis = np.array([0, 0, 1])
+ axis = np.cross(z_axis, vec)
+ if np.linalg.norm(axis) > 1e-6:
+ axis = axis / np.linalg.norm(axis)
+ angle = np.arccos(np.dot(z_axis, vec) / height)
+ rot = trimesh.transformations.rotation_matrix(angle, axis)
+ tm.apply_transform(rot)
+
+ tm.apply_translation(start + vec / 2)
+
+ # Material color (supports RGBA)
+ rgba = np.array(color)
+ material = pyrender.MetallicRoughnessMaterial(
+ metallicFactor=0.1,
+ roughnessFactor=0.5,
+ baseColorFactor=rgba
+ )
+
+ mesh = pyrender.Mesh.from_trimesh(tm, material=material)
+ scene.add(mesh)
+
+ # Projected points for visualization, check if projection is correct
+ x1 = fx * (start[0] / start[2]) + cx
+ y1 = fy * (start[1] / start[2]) + cy
+ x2 = fx * (end[0] / end[2]) + cx
+ y2 = fy * (end[1] / end[2]) + cy
+ points_to_draw.append((x1, y1))
+ points_to_draw.append((x2, y2))
+
+
+ # Render
+ r = pyrender.OffscreenRenderer(viewport_width=W, viewport_height=H, point_size=1.0)
+ color, _ = r.render(scene, flags=pyrender.RenderFlags.RGBA)
+
+ # Post-processing
+ color = color.astype(np.float32) / 255.0
+ final_img = (color * 255).astype(np.uint8)
+
+ # Draw points, check if projection is correct
+ for (x, y) in points_to_draw:
+ print(f" debug point: {x}, {y}")
+ x_draw = int(x)
+ y_draw = int(y)
+ cv2.circle(final_img, (x_draw, y_draw), radius=4, color=(0, 255, 0), thickness=-1)
+
+ return Image.fromarray(final_img)
+
diff --git a/render_3d/taichi_cylinder.py b/render_3d/taichi_cylinder.py
new file mode 100644
index 0000000..1c99210
--- /dev/null
+++ b/render_3d/taichi_cylinder.py
@@ -0,0 +1,207 @@
+import taichi as ti
+import numpy as np
+import random
+import math
+
+ti.init(arch=ti.cuda)
+
+def flatten_specs(specs_list):
+ """把 specs_list 拉平为 numpy 数组 + 索引表"""
+ starts, ends, colors = [], [], []
+ frame_offset, frame_count = [], []
+ offset = 0
+ for specs in specs_list:
+ frame_offset.append(offset)
+ frame_count.append(len(specs))
+ for (s, e, c) in specs:
+ starts.append(s)
+ ends.append(e)
+ colors.append(c)
+ offset += len(specs)
+ return (
+ np.array(starts, dtype=np.float32),
+ np.array(ends, dtype=np.float32),
+ np.array(colors, dtype=np.float32),
+ np.array(frame_offset, dtype=np.int32),
+ np.array(frame_count, dtype=np.int32),
+ )
+
+def render_whole(specs_list, H=480, W=640, fx=500, fy=500, cx=240, cy=320, radius=21.5):
+ img = ti.Vector.field(4, dtype=ti.f32, shape=(H, W))
+ starts, ends, colors, frame_offset, frame_count = flatten_specs(specs_list)
+ total_cyl = len(starts)
+ n_frames = len(specs_list)
+ z_min = min(starts[:, 2].min(), ends[:, 2].min())
+ z_max = max(starts[:, 2].max(), ends[:, 2].max())
+
+ # ========= 相机内参 =========
+ znear = 0.1
+ zfar = max(min(z_max, 25000), 10000)
+ C = ti.Vector([0.0, 0.0, 0.0]) # 相机中心
+ light_dir = ti.Vector([0.0, 0.0, 1.0])
+
+ c_start = ti.Vector.field(3, dtype=ti.f32, shape=total_cyl)
+ c_end = ti.Vector.field(3, dtype=ti.f32, shape=total_cyl)
+ c_rgba = ti.Vector.field(4, dtype=ti.f32, shape=total_cyl)
+ n_cyl = ti.field(dtype=ti.i32, shape=()) # 实际数量
+ f_offset = ti.field(dtype=ti.i32, shape=n_frames)
+ f_count = ti.field(dtype=ti.i32, shape=n_frames)
+ frame_id = ti.field(dtype=ti.i32, shape=()) # 当前帧号
+ z_min_field = ti.field(dtype=ti.f32, shape=())
+ z_max_field = ti.field(dtype=ti.f32, shape=())
+
+ z_min_field[None] = z_min
+ z_max_field[None] = z_max
+
+ # # ====== 拷贝数据一次 ======
+ c_start.from_numpy(starts)
+ c_end.from_numpy(ends)
+ c_rgba.from_numpy(colors)
+ f_offset.from_numpy(frame_offset)
+ f_count.from_numpy(frame_count)
+
+ @ti.func
+ def sd_cylinder(p, a, b, r):
+ pa = p - a
+ ba = b - a
+ h = ba.norm()
+ eps = 1e-8
+ res = 0.0
+ if h < eps:
+ res = pa.norm() - r
+ else:
+ ba_n = ba / h
+ proj = pa.dot(ba_n)
+ proj_clamped = min(max(proj, 0.0), h)
+ res = (pa - proj_clamped * ba_n).norm() - r
+ return res
+
+ @ti.func
+ def scene_sdf(p):
+ best_d = 1e6
+ best_col = ti.Vector([0.0, 0.0, 0.0, 0.0])
+ fid = frame_id[None] # 从 field 里读出来,变成一个普通 int
+ off = f_offset[fid]
+ cnt = f_count[fid]
+ for i in range(cnt): # 只遍历实际数量
+ a = c_start[off + i]
+ b = c_end[off + i]
+ r = radius
+ col = c_rgba[off + i]
+ d = sd_cylinder(p, a, b, r)
+ if d < best_d:
+ best_d = d
+ best_col = col
+ return best_d, best_col
+
+ @ti.func
+ def get_normal(p):
+ e = 1e-3
+ dx = scene_sdf(p + ti.Vector([e, 0.0, 0.0]))[0] - scene_sdf(p - ti.Vector([e, 0.0, 0.0]))[0]
+ dy = scene_sdf(p + ti.Vector([0.0, e, 0.0]))[0] - scene_sdf(p - ti.Vector([0.0, e, 0.0]))[0]
+ dz = scene_sdf(p + ti.Vector([0.0, 0.0, e]))[0] - scene_sdf(p - ti.Vector([0.0, 0.0, e]))[0]
+ n = ti.Vector([dx, dy, dz])
+ return n.normalized()
+
+ @ti.func
+ def pixel_to_ray(xi, yi):
+ u = (xi - cx) / fx
+ v = (yi - cy) / fy
+ dir_cam = ti.Vector([u, v, 1.0]).normalized()
+ Rcw = ti.Matrix.identity(ti.f32, 3)
+ rd_world = Rcw @ dir_cam
+ ro_world = C
+ return ro_world, rd_world
+
+ @ti.kernel
+ def render():
+ depth_near, depth_far = ti.max(z_min_field[None], 0.1), ti.min(z_max_field[None] + 6000, 20000) # 能渲染出来的点,最大12000
+ for y, x in img:
+ ro, rd = pixel_to_ray(x, y)
+ t = znear
+ col_out = ti.Vector([0.0, 0.0, 0.0, 0.0])
+ for _ in range(300):
+ p = ro + rd * t
+ d, col = scene_sdf(p)
+ if d < 1e-3:
+ # n = get_normal(p)
+ # diff = max(n.dot(-light_dir), 0.0)
+ # lit = 0.3 + 0.7 * diff
+ # col_out = ti.Vector([col.x * lit, col.y * lit, col.z * lit, col.w])
+ # break
+
+ n = get_normal(p)
+ diff = max(n.dot(-light_dir), 0.0)
+
+ # === Blinn-Phong 镜面反射 ===
+ view_dir = -rd.normalized()
+ half_dir = (view_dir + -light_dir).normalized()
+ spec = max(n.dot(half_dir), 0.0) ** 32 # shininess=32,越小越散,越大越锐
+
+ depth_factor = 1.0 - (p.z - depth_near) / (depth_far - znear)
+ depth_factor = ti.max(0.0, ti.min(1.0, depth_factor))
+
+ # 原来的 diffuse/ambient 光照
+ diffuse_term = 0.3 + 0.7 * diff
+ base = col.xyz * diffuse_term * depth_factor
+
+ # 镜面高光(叠加到原有结果上)
+ highlight = ti.Vector([1.0, 1.0, 1.0]) * (0.5 * spec) * depth_factor
+
+ col_out = ti.Vector([base.x + highlight.x,
+ base.y + highlight.y,
+ base.z + highlight.z,
+ col.w])
+ break
+
+ if t > zfar:
+ break
+ t += max(d, 1e-4)
+ img[y, x] = col_out
+
+ frames_np_rgba = []
+ for f in range(len(specs_list)):
+ # start_time = time.time()
+ frame_id[None] = f
+ render()
+ arr = np.clip(img.to_numpy(), 0, 1)
+ # end_time = time.time()
+ # print(f"Frame {f} time: {end_time - start_time} seconds")
+ arr8 = (arr * 255).astype(np.uint8)
+ frames_np_rgba.append(arr8)
+
+ return frames_np_rgba
+
+
+def random_cylinder():
+ """生成一根随机圆柱 (start, end, color)。"""
+ # 起点 [-200,200]^2, z 在 [-300,-100]
+ ax = random.uniform(-200, 200)
+ ay = random.uniform(-200, 200)
+ az = random.uniform(300, 400)
+ start = [ax, ay, az]
+
+ # 随机方向和长度
+ theta = random.uniform(0, 2*math.pi)
+ phi = random.uniform(-math.pi/4, math.pi/4) # 倾斜角
+ L = 100
+ dx = math.cos(phi) * math.cos(theta)
+ dy = math.cos(phi) * math.sin(theta)
+ dz = math.sin(phi)
+ end = [ax + dx * L, ay + dy * L, az + dz * L]
+
+ # 随机颜色 (RGB + alpha=1)
+ color = [random.random(), random.random(), random.random(), 1.0]
+
+ return (start, end, color)
+
+def generate_specs_list(num_frames=120, min_cyl=10, max_cyl=120):
+ """生成 specs_list,每帧有若干随机圆柱."""
+ specs_list = []
+ for _ in range(num_frames):
+ n_cyl = random.randint(min_cyl, max_cyl)
+ specs = [random_cylinder() for _ in range(n_cyl)]
+ specs_x_shift = [([spec[0][0] + 50, spec[0][1], spec[0][2]], [spec[1][0] + 50, spec[1][1], spec[1][2]], spec[2]) for spec in specs]
+ specs_list.append(specs)
+ specs_list.append(specs_x_shift)
+ return specs_list
diff --git a/requirements.txt b/requirements.txt
new file mode 100644
index 0000000..7f0ba37
--- /dev/null
+++ b/requirements.txt
@@ -0,0 +1,6 @@
+taichi
+pyrender
+trimesh
+opencv-python
+matplotlib
+pillow
\ No newline at end of file
diff --git a/vitpose_utils/utils.py b/vitpose_utils/utils.py
new file mode 100644
index 0000000..1a88b4e
--- /dev/null
+++ b/vitpose_utils/utils.py
@@ -0,0 +1,191 @@
+
+
+import numpy as np
+import cv2
+
+# source
+# https://github.com/Wan-Video/Wan2.2/blob/e9783574ef77be11fcab9aa5607905402538c08d/wan/modules/animate/preprocess/pose2d_utils.py#L1034
+
+def bbox_from_detector(bbox, input_resolution=(224, 224), rescale=1.25):
+ """
+ Get center and scale of bounding box from bounding box.
+ The expected format is [min_x, min_y, max_x, max_y].
+ """
+ CROP_IMG_HEIGHT, CROP_IMG_WIDTH = input_resolution
+ CROP_ASPECT_RATIO = CROP_IMG_HEIGHT / float(CROP_IMG_WIDTH)
+
+ # center
+ center_x = (bbox[0] + bbox[2]) / 2.0
+ center_y = (bbox[1] + bbox[3]) / 2.0
+ center = np.array([center_x, center_y])
+
+ # scale
+ bbox_w = bbox[2] - bbox[0]
+ bbox_h = bbox[3] - bbox[1]
+ bbox_size = max(bbox_w * CROP_ASPECT_RATIO, bbox_h)
+
+ scale = np.array([bbox_size / CROP_ASPECT_RATIO, bbox_size]) / 200.0
+ # scale = bbox_size / 200.0
+ # adjust bounding box tightness
+ scale *= rescale
+ return center, scale
+
+def get_transform(center, scale, res, rot=0):
+ """Generate transformation matrix."""
+ # res: (height, width), (rows, cols)
+ crop_aspect_ratio = res[0] / float(res[1])
+ h = 200 * scale
+ w = h / crop_aspect_ratio
+ t = np.zeros((3, 3))
+ t[0, 0] = float(res[1]) / w
+ t[1, 1] = float(res[0]) / h
+ t[0, 2] = res[1] * (-float(center[0]) / w + .5)
+ t[1, 2] = res[0] * (-float(center[1]) / h + .5)
+ t[2, 2] = 1
+ if not rot == 0:
+ rot = -rot # To match direction of rotation from cropping
+ rot_mat = np.zeros((3, 3))
+ rot_rad = rot * np.pi / 180
+ sn, cs = np.sin(rot_rad), np.cos(rot_rad)
+ rot_mat[0, :2] = [cs, -sn]
+ rot_mat[1, :2] = [sn, cs]
+ rot_mat[2, 2] = 1
+ # Need to rotate around center
+ t_mat = np.eye(3)
+ t_mat[0, 2] = -res[1] / 2
+ t_mat[1, 2] = -res[0] / 2
+ t_inv = t_mat.copy()
+ t_inv[:2, 2] *= -1
+ t = np.dot(t_inv, np.dot(rot_mat, np.dot(t_mat, t)))
+ return t
+
+def transform(pt, center, scale, res, invert=0, rot=0):
+ """Transform pixel location to different reference."""
+ t = get_transform(center, scale, res, rot=rot)
+ if invert:
+ t = np.linalg.inv(t)
+ new_pt = np.array([pt[0] - 1, pt[1] - 1, 1.]).T
+ new_pt = np.dot(t, new_pt)
+ return np.array([round(new_pt[0]), round(new_pt[1])], dtype=int) + 1
+
+def crop(img, center, scale, res):
+ """
+ Crop image according to the supplied bounding box.
+ res: [rows, cols]
+ """
+ # Upper left point
+ ul = np.array(transform([1, 1], center, max(scale), res, invert=1)) - 1
+ # Bottom right point
+ br = np.array(transform([res[1] + 1, res[0] + 1], center, max(scale), res, invert=1)) - 1
+
+ new_shape = [br[1] - ul[1], br[0] - ul[0]]
+ if len(img.shape) > 2:
+ new_shape += [img.shape[2]]
+ new_img = np.zeros(new_shape, dtype=np.float32)
+
+ # Range to fill new array
+ new_x = max(0, -ul[0]), min(br[0], len(img[0])) - ul[0]
+ new_y = max(0, -ul[1]), min(br[1], len(img)) - ul[1]
+ # Range to sample from original image
+ old_x = max(0, ul[0]), min(len(img[0]), br[0])
+ old_y = max(0, ul[1]), min(len(img), br[1])
+ try:
+ new_img[new_y[0]:new_y[1], new_x[0]:new_x[1]] = img[old_y[0]:old_y[1], old_x[0]:old_x[1]]
+ except Exception as e:
+ print(e)
+
+ new_img = cv2.resize(new_img, (res[1], res[0])) # (cols, rows)
+ return new_img, new_shape, (old_x, old_y), (new_x, new_y) # , ul, br
+
+
+def split_kp2ds_for_aa(kp2ds, ret_face=False):
+ kp2ds_body = (kp2ds[[0, 6, 6, 8, 10, 5, 7, 9, 12, 14, 16, 11, 13, 15, 2, 1, 4, 3, 17, 20]] + kp2ds[[0, 5, 6, 8, 10, 5, 7, 9, 12, 14, 16, 11, 13, 15, 2, 1, 4, 3, 18, 21]]) / 2
+ kp2ds_lhand = kp2ds[91:112]
+ kp2ds_rhand = kp2ds[112:133]
+ kp2ds_face = kp2ds[22:91]
+ if ret_face:
+ return kp2ds_body.copy(), kp2ds_lhand.copy(), kp2ds_rhand.copy(), kp2ds_face.copy()
+ return kp2ds_body.copy(), kp2ds_lhand.copy(), kp2ds_rhand.copy()
+
+
+def load_pose_metas_from_kp2ds_seq(kp2ds_seq, width, height):
+ metas = []
+ last_kp2ds_body = None
+ for kps in kp2ds_seq:
+ kps = kps.copy()
+ kps[:, 0] /= width
+ kps[:, 1] /= height
+ kp2ds_body, kp2ds_lhand, kp2ds_rhand, kp2ds_face = split_kp2ds_for_aa(kps, ret_face=True)
+
+ # Exclude cases where all values are less than 0
+ if last_kp2ds_body is not None and kp2ds_body[:, :2].min(axis=1).max() < 0:
+ kp2ds_body = last_kp2ds_body
+ last_kp2ds_body = kp2ds_body
+
+ meta = {
+ "width": width,
+ "height": height,
+ "keypoints_body": kp2ds_body,
+ "keypoints_left_hand": kp2ds_lhand,
+ "keypoints_right_hand": kp2ds_rhand,
+ "keypoints_face": kp2ds_face,
+ }
+ metas.append(meta)
+ return metas
+
+def aaposemeta_to_dwpose_scail(meta):
+ """
+ Convert AA pose metadata to DWpose format matching DWposeDetector output.
+
+ DWpose format:
+ - bodies: dict with 'candidate' (n, 24, 2) and 'subset' (n, 24) where subset contains indices
+ - hands: array (2*n, 21, 2) - stacked right/left hands
+ - faces: array (n, 68, 2)
+ """
+ # Body keypoints (excluding last 2)
+ candidate_body = meta['keypoints_body'][:-2][:, :2] # (24, 2)
+ score_body = meta['keypoints_body'][:-2][:, 2] # (24,)
+
+ # Create subset: contains joint index if visible, -1 if not
+ subset_body = np.arange(len(candidate_body), dtype=float)
+ subset_body[score_body <= 0.3] = -1 # Match DWpose threshold
+
+ # Bodies dict with single person (expand to match multi-person format)
+ bodies = {
+ "candidate": np.expand_dims(candidate_body, axis=0), # (1, 24, 2)
+ "subset": np.expand_dims(subset_body, axis=0) # (1, 24)
+ }
+
+ # Hands: stack right then left (2, 21, 2)
+ hands_coords = np.stack([
+ meta['keypoints_right_hand'][:, :2],
+ meta['keypoints_left_hand'][:, :2]
+ ], axis=0)
+
+ hands_score = np.stack([
+ meta['keypoints_right_hand'][:, 2],
+ meta['keypoints_left_hand'][:, 2]
+ ], axis=0)
+
+ # Faces: (1, 68, 2) - skip first face keypoint like DWpose does (24:92 = 68 points)
+ faces_coords = np.expand_dims(meta['keypoints_face'][1:][:, :2], axis=0)
+ faces_score = np.expand_dims(meta['keypoints_face'][1:][:, 2], axis=0)
+
+ # Match DWpose output structure
+ dwpose_format = {
+ "bodies": bodies,
+ "hands": hands_coords,
+ "faces": faces_coords
+ }
+
+ # Optional: include scores separately like DWpose does
+ score_dict = {
+ "body_score": np.expand_dims(score_body, axis=0),
+ "hand_score": hands_score,
+ "face_score": faces_score
+ }
+
+ # Merge score dict into dwpose_format
+ dwpose_format.update(score_dict)
+
+ return dwpose_format