From 93059f0fe4ea3881c0d6ec0e327e5d920554277d Mon Sep 17 00:00:00 2001 From: kijai <40791699+kijai@users.noreply.github.com> Date: Sun, 14 Dec 2025 19:16:28 +0200 Subject: [PATCH] Init --- .gitignore | 13 + NLFPoseExtract/align3d.py | 123 +++ NLFPoseExtract/nlf_render.py | 400 +++++++++ __init__.py | 3 + .../SCAIL_preprocess_example_01.json | 825 ++++++++++++++++++ nodes.py | 243 ++++++ pose_draw/draw_3d_utils.py | 221 +++++ pose_draw/draw_pose_utils.py | 129 +++ pose_draw/draw_utils.py | 658 ++++++++++++++ pyproject.toml | 15 + readme.md | 12 + render_3d/render_cylinder.py | 93 ++ render_3d/taichi_cylinder.py | 207 +++++ requirements.txt | 6 + vitpose_utils/utils.py | 191 ++++ 15 files changed, 3139 insertions(+) create mode 100644 .gitignore create mode 100644 NLFPoseExtract/align3d.py create mode 100644 NLFPoseExtract/nlf_render.py create mode 100644 __init__.py create mode 100644 example_workflows/SCAIL_preprocess_example_01.json create mode 100644 nodes.py create mode 100644 pose_draw/draw_3d_utils.py create mode 100644 pose_draw/draw_pose_utils.py create mode 100644 pose_draw/draw_utils.py create mode 100644 pyproject.toml create mode 100644 readme.md create mode 100644 render_3d/render_cylinder.py create mode 100644 render_3d/taichi_cylinder.py create mode 100644 requirements.txt create mode 100644 vitpose_utils/utils.py diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..d32eacc --- /dev/null +++ b/.gitignore @@ -0,0 +1,13 @@ +output/ +*__pycache__/ +samples*/ +runs/ +checkpoints/ +master_ip +logs/ +*.DS_Store +.idea +tools/ +.vscode/ +convert_* +*.pt \ No newline at end of file diff --git a/NLFPoseExtract/align3d.py b/NLFPoseExtract/align3d.py new file mode 100644 index 0000000..20e6e06 --- /dev/null +++ b/NLFPoseExtract/align3d.py @@ -0,0 +1,123 @@ +import numpy as np +from scipy.optimize import minimize + + +def solve_new_camera_params_central(three_d_points, focal_length, imshape, new_2d_points): + """ + Solve for new camera parameters by minimizing the error between the original 2D projection points and the new 2D projection points. + + Args: + three_d_points (torch.Tensor): N*3 3D points + focal_length (float): Focal length of the original camera + imshape (tuple): Image size, e.g., [512, 896] + original_2d_points (torch.Tensor): N*2 original 2D projection points + new_2d_points (torch.Tensor): N*2 new 2D projection points + + Returns: + m, n, p, q: Parameters in the new camera intrinsic matrix + """ + + + # Objective function: minimize the error between the original projection points and the new projection points + def objective(params): + m, s, p, q = params + # Construct the new camera intrinsic matrix + K_new = np.array([ + [focal_length * m , 0, imshape[1] / 2 + p], + [0, focal_length * m * s, imshape[0] / 2 + q], + [0, 0, 1] + ]) + + # Compute the new 2D projection points + new_projections = [] + for point in three_d_points: + X, Y, Z = point + u = (K_new[0, 0] * X / Z) + K_new[0, 2] + v = (K_new[1, 1] * Y / Z) + K_new[1, 2] + new_projections.append([u, v]) + new_projections = np.array(new_projections) + + # Calculate the error between the original 2D projection points and the new projection points + # Special handling for the 0th projection point + error0 = np.sum((new_2d_points[:1] - new_projections[:1]) ** 2) + error = np.sum((new_2d_points[1:] - new_projections[1:]) ** 2) + return error0 * 8 + error + + # Initialize parameters m, beta, p, q + initial_params = [1.0, 1.0, 0.0, 0.0] # Initial values + + # Use least squares to solve for p, q + result = minimize(objective, initial_params, bounds=[(0.7, 1.4), (0.8, 1.15), (-imshape[1], imshape[1]), (-imshape[0], imshape[0])]) + + # Output the solution result + m, s, p, q = result.x + print(f"debug: solved camera params m={m}, s={s}, p={p}, q={q}") + + K_final = np.array([ + [focal_length * m, 0, imshape[1] / 2 + p], + [0, focal_length * m * s, imshape[0] / 2 + q], + [0, 0, 1] + ]) + + + return K_final, m + + +def solve_new_camera_params_down(three_d_points, focal_length, imshape, new_2d_points): + """ + Solve for new camera parameters by minimizing the error between the original 2D projection points and the new 2D projection points. + + Args: + three_d_points (torch.Tensor): N*3 3D points + focal_length (float): Focal length of the original camera + imshape (tuple): Image size, e.g., [512, 896] + original_2d_points (torch.Tensor): N*2 original 2D projection points + new_2d_points (torch.Tensor): N*2 new 2D projection points + + Returns: + m, n, p, q: Parameters in the new camera intrinsic matrix + """ + + # Objective function: minimize the error between the original projection points and the new projection points + def objective(params): + m, s, p, q = params + # Construct the new camera intrinsic matrix + K_new = np.array([ + [focal_length * m , 0, imshape[1] / 2 + p], + [0, focal_length * m * s, imshape[0] / 2 + q], + [0, 0, 1] + ]) + + # Compute the new 2D projection points + new_projections = [] + for point in three_d_points: + X, Y, Z = point + u = (K_new[0, 0] * X / Z) + K_new[0, 2] + v = (K_new[1, 1] * Y / Z) + K_new[1, 2] + new_projections.append([u, v]) + new_projections = np.array(new_projections) + + # Calculate the error between the original 2D projection points and the new projection points + # Special handling for the 0th projection point + error0 = np.sum((new_2d_points[:1] - new_projections[:1]) ** 2) + error = np.sum((new_2d_points[1:] - new_projections[1:]) ** 2) + return error0 + error * 4 + + # Initialize parameters m, beta, p, q + initial_params = [1.0, 1.0, 0.0, 0.0] # Initial values + + # Use least squares to solve for p, q + result = minimize(objective, initial_params, bounds=[(0.7, 1.4), (0.8, 1.15), (-imshape[1], imshape[1]), (-imshape[0], imshape[0])]) + + # Output the solution result + m, s, p, q = result.x + print(f"debug: solved camera params m={m}, s={s}, p={p}, q={q}") + + K_final = np.array([ + [focal_length * m, 0, imshape[1] / 2 + p], + [0, focal_length * m * s, imshape[0] / 2 + q], + [0, 0, 1] + ]) + + + return K_final, m diff --git a/NLFPoseExtract/nlf_render.py b/NLFPoseExtract/nlf_render.py new file mode 100644 index 0000000..9abb43c --- /dev/null +++ b/NLFPoseExtract/nlf_render.py @@ -0,0 +1,400 @@ +import numpy as np +import torch +import os, platform, copy + +if platform.system() == 'Linux': + if 'PYOPENGL_PLATFORM' not in os.environ: + os.environ['PYOPENGL_PLATFORM'] = 'egl' +elif platform.system() == 'Windows': + os.environ.pop('PYOPENGL_PLATFORM', None) + +from ..render_3d.taichi_cylinder import render_whole +from ..pose_draw.draw_pose_utils import draw_pose_to_canvas_np + +def p3d_single_p2d(points, intrinsic_matrix): + X, Y, Z = points[0], points[1], points[2] + u = (intrinsic_matrix[0, 0] * X / Z) + intrinsic_matrix[0, 2] + v = (intrinsic_matrix[1, 1] * Y / Z) + intrinsic_matrix[1, 2] + u_np = u.cpu().numpy() + v_np = v.cpu().numpy() + return np.array([u_np, v_np]) + +def process_data_to_COCO_format(joints): + """Args: + joints: numpy array of shape (24, 2) or (24, 3) + Returns: + new_joints: numpy array of shape (17, 2) or (17, 3) + """ + if joints.ndim != 2: + raise ValueError(f"Expected shape (24,2) or (24,3), got {joints.shape}") + + dim = joints.shape[1] # 2D or 3D + + mapping = { + 15: 0, # head + 12: 1, # neck + 17: 2, # left shoulder + 16: 5, # right shoulder + 19: 3, # left elbow + 18: 6, # right elbow + 21: 4, # left hand + 20: 7, # right hand + 2: 8, # left pelvis + 1: 11, # right pelvis + 5: 9, # left knee + 4: 12, # right knee + 8: 10, # left feet + 7: 13, # right feet + } + + new_joints = np.zeros((18, dim), dtype=joints.dtype) + for src, dst in mapping.items(): + new_joints[dst] = joints[src] + + return new_joints + +def intrinsic_matrix_from_field_of_view(imshape, fov_degrees:float =55): # nlf default fov_degrees 55 + imshape = np.array(imshape) + fov_radians = fov_degrees * np.array(np.pi / 180) + larger_side = np.max(imshape) + focal_length = larger_side / (np.tan(fov_radians / 2) * 2) + # intrinsic_matrix 3*3 + return np.array([ + [focal_length, 0, imshape[1] / 2], + [0, focal_length, imshape[0] / 2], + [0, 0, 1], + ]) + +def shift_dwpose_according_to_nlf(smpl_poses, aligned_poses, ori_intrinstics, modified_intrinstics, height, width): + ########## warning: 会改变body; shift 之后 body是不准的 ########## + for i in range(len(smpl_poses)): + persons_joints_list = smpl_poses[i] + poses_list = aligned_poses[i] + # 对里面每一个人,取关节并进行变形;并且修改2d;如果3d不存在,把2d的手/脸也去掉 + for person_idx, person_joints in enumerate(persons_joints_list): + face = poses_list["faces"][person_idx] + right_hand = poses_list["hands"][2 * person_idx] + left_hand = poses_list["hands"][2 * person_idx + 1] + candidate = poses_list["bodies"]["candidate"][person_idx] + # 注意,这里不是coco format + person_joint_15_2d_shift = p3d_single_p2d(person_joints[15], modified_intrinstics) - p3d_single_p2d(person_joints[15], ori_intrinstics) if person_joints[15, 2] > 0.01 else np.array([0.0, 0.0]) # face + person_joint_20_2d_shift = p3d_single_p2d(person_joints[20], modified_intrinstics) - p3d_single_p2d(person_joints[20], ori_intrinstics) if person_joints[20, 2] > 0.01 else np.array([0.0, 0.0]) # right hand + person_joint_21_2d_shift = p3d_single_p2d(person_joints[21], modified_intrinstics) - p3d_single_p2d(person_joints[21], ori_intrinstics) if person_joints[21, 2] > 0.01 else np.array([0.0, 0.0]) # left hand + + face[:, 0] += person_joint_15_2d_shift[0] / width + face[:, 1] += person_joint_15_2d_shift[1] / height + right_hand[:, 0] += person_joint_20_2d_shift[0] / width + right_hand[:, 1] += person_joint_20_2d_shift[1] / height + left_hand[:, 0] += person_joint_21_2d_shift[0] / width + left_hand[:, 1] += person_joint_21_2d_shift[1] / height + candidate[:, 0] += person_joint_15_2d_shift[0] / width + candidate[:, 1] += person_joint_15_2d_shift[1] / height + + +def get_single_pose_cylinder_specs(args): + """Helper function for rendering a single pose, used for parallel processing.""" + idx, pose, focal, princpt, height, width, colors, limb_seq, draw_seq = args + cylinder_specs = [] + + for joints3d in pose: # 多人 + joints3d = joints3d.cpu().numpy() + joints3d = process_data_to_COCO_format(joints3d) + for line_idx in draw_seq: + line = limb_seq[line_idx] + start, end = line[0], line[1] + if np.sum(joints3d[start]) == 0 or np.sum(joints3d[end]) == 0: + continue + else: + cylinder_specs.append((joints3d[start], joints3d[end], colors[line_idx])) + return cylinder_specs + + +def collect_smpl_poses(data): + uncollected_smpl_poses = [item['nlfpose'] for item in data] + smpl_poses = [[] for _ in range(len(uncollected_smpl_poses))] + for frame_idx in range(len(uncollected_smpl_poses)): + for person_idx in range(len(uncollected_smpl_poses[frame_idx])): # 每个人(每个bbox)只给出一个pose + if len(uncollected_smpl_poses[frame_idx][person_idx]) > 0: # 有返回的骨骼 + smpl_poses[frame_idx].append(uncollected_smpl_poses[frame_idx][person_idx][0]) + else: + smpl_poses[frame_idx].append(torch.zeros((24, 3), dtype=torch.float32)) # 没有检测到人,就放一个全0的 + + return smpl_poses + + + +def collect_smpl_poses_samurai(data): + uncollected_smpl_poses = [item['nlfpose'] for item in data] + smpl_poses_first = [[] for _ in range(len(uncollected_smpl_poses))] + smpl_poses_second = [[] for _ in range(len(uncollected_smpl_poses))] + + for frame_idx in range(len(uncollected_smpl_poses)): + for person_idx in range(len(uncollected_smpl_poses[frame_idx])): # 每个人(每个bbox)只给出一个pose + if len(uncollected_smpl_poses[frame_idx][person_idx]) > 0: # 有返回的骨骼 + if person_idx == 0: + smpl_poses_first[frame_idx].append(uncollected_smpl_poses[frame_idx][person_idx][0]) + elif person_idx == 1: + smpl_poses_second[frame_idx].append(uncollected_smpl_poses[frame_idx][person_idx][0]) + else: + if person_idx == 0: + smpl_poses_first[frame_idx].append(torch.zeros((24, 3), dtype=torch.float32)) # 没有检测到人,就放一个全0的 + elif person_idx == 1: + smpl_poses_second[frame_idx].append(torch.zeros((24, 3), dtype=torch.float32)) + + return smpl_poses_first, smpl_poses_second + + + + +def render_nlf_as_images(smpl_poses, dw_poses, height, width, video_length, intrinsic_matrix=None, draw_2d=True): + """ return a list of images """ + + base_colors_255_dict = { + # Warm Colors for Right Side (R.) - Red, Orange, Yellow + "Red": [255, 0, 0], + "Orange": [255, 85, 0], + "Golden Orange": [255, 170, 0], + "Yellow": [255, 240, 0], + "Yellow-Green": [180, 255, 0], + # Cool Colors for Left Side (L.) - Green, Blue, Purple + "Bright Green": [0, 255, 0], + "Light Green-Blue": [0, 255, 85], + "Aqua": [0, 255, 170], + "Cyan": [0, 255, 255], + "Sky Blue": [0, 170, 255], + "Medium Blue": [0, 85, 255], + "Pure Blue": [0, 0, 255], + "Purple-Blue": [85, 0, 255], + "Medium Purple": [170, 0, 255], + # Neutral/Central Colors (e.g., for Neck, Nose, Eyes, Ears) + "Grey": [150, 150, 150], + "Pink-Magenta": [255, 0, 170], + "Dark Pink": [255, 0, 85], + "Violet": [100, 0, 255], + "Dark Violet": [50, 0, 255], + } + + ordered_colors_255 = [ + base_colors_255_dict["Red"], # Neck -> R. Shoulder (Red) + base_colors_255_dict["Cyan"], # Neck -> L. Shoulder (Cyan) + base_colors_255_dict["Orange"], # R. Shoulder -> R. Elbow (Orange) + base_colors_255_dict["Golden Orange"], # R. Elbow -> R. Wrist (Golden Orange) + base_colors_255_dict["Sky Blue"], # L. Shoulder -> L. Elbow (Sky Blue) + base_colors_255_dict["Medium Blue"], # L. Elbow -> L. Wrist (Medium Blue) + base_colors_255_dict["Yellow-Green"], # Neck -> R. Hip ( Yellow-Green) + base_colors_255_dict["Bright Green"], # R. Hip -> R. Knee (Bright Green - transitioning warm to cool spectrum) + base_colors_255_dict["Light Green-Blue"], # R. Knee -> R. Ankle (Light Green-Blue - transitioning) + base_colors_255_dict["Pure Blue"], # Neck -> L. Hip (Pure Blue) + base_colors_255_dict["Purple-Blue"], # L. Hip -> L. Knee (Purple-Blue) + base_colors_255_dict["Medium Purple"], # L. Knee -> L. Ankle (Medium Purple) + base_colors_255_dict["Grey"], # Neck -> Nose (Grey) + base_colors_255_dict["Pink-Magenta"], # Nose -> R. Eye (Pink/Magenta) + base_colors_255_dict["Dark Violet"], # R. Eye -> R. Ear (Dark Pink) + base_colors_255_dict["Pink-Magenta"], # Nose -> L. Eye (Violet) + base_colors_255_dict["Dark Violet"], # L. Eye -> L. Ear (Dark Violet) + ] + + limb_seq = [ + [1, 2], # 0 Neck -> R. Shoulder + [1, 5], # 1 Neck -> L. Shoulder + [2, 3], # 2 R. Shoulder -> R. Elbow + [3, 4], # 3 R. Elbow -> R. Wrist + [5, 6], # 4 L. Shoulder -> L. Elbow + [6, 7], # 5 L. Elbow -> L. Wrist + [1, 8], # 6 Neck -> R. Hip + [8, 9], # 7 R. Hip -> R. Knee + [9, 10], # 8 R. Knee -> R. Ankle + [1, 11], # 9 Neck -> L. Hip + [11, 12], # 10 L. Hip -> L. Knee + [12, 13], # 11 L. Knee -> L. Ankle + [1, 0], # 12 Neck -> Nose + [0, 14], # 13 Nose -> R. Eye + [14, 16], # 14 R. Eye -> R. Ear + [0, 15], # 15 Nose -> L. Eye + [15, 17], # 16 L. Eye -> L. Ear + ] + + draw_seq = [0, 2, 3, # Neck -> R. Shoulder -> R. Elbow -> R. Wrist + 1, 4, 5, # Neck -> L. Shoulder -> L. Elbow -> L. Wrist + 6, 7, 8, # Neck -> R. Hip -> R. Knee -> R. Ankle + 9, 10, 11, # Neck -> L. Hip -> L. Knee -> L. Ankle + 12, # Neck -> Nose + 13, 14, # Nose -> R. Eye -> R. Ear + 15, 16, # Nose -> L. Eye -> L. Ear + ] # Expanding outward from the proximal end + + colors = [[c / 300 + 0.15 for c in color_rgb] + [0.8] for color_rgb in ordered_colors_255] + + if dw_poses is not None: + aligned_poses = copy.deepcopy(dw_poses) + + if intrinsic_matrix is None: + intrinsic_matrix = intrinsic_matrix_from_field_of_view((height, width)) + focal_x = intrinsic_matrix[0,0] + focal_y = intrinsic_matrix[1,1] + princpt = (intrinsic_matrix[0,2], intrinsic_matrix[1,2]) # (cx, cy) + + + # obtain cylinder_specs for each frame + cylinder_specs_list = [] + for i in range(video_length): + cylinder_specs = get_single_pose_cylinder_specs((i, smpl_poses[i], None, None, None, None, colors, limb_seq, draw_seq)) + cylinder_specs_list.append(cylinder_specs) + + + frames_np_rgba = render_whole(cylinder_specs_list, H=height, W=width, fx=focal_x, fy=focal_y, cx=princpt[0], cy=princpt[1]) + if dw_poses is not None and draw_2d: + canvas_2d = draw_pose_to_canvas_np(aligned_poses, pool=None, H=height, W=width, reshape_scale=0, show_feet_flag=False, show_body_flag=False, show_cheek_flag=True, dw_hand=True) + + for i in range(len(frames_np_rgba)): + frame_img = frames_np_rgba[i] + canvas_img = canvas_2d[i] + mask = canvas_img != 0 + frame_img[:, :, :3][mask] = canvas_img[mask] + frames_np_rgba[i] = frame_img + + return frames_np_rgba + + +def render_multi_nlf_as_images(data, dw_poses, intrinsic_matrix=None, draw_2d=True): + """ return a list of images """ + height, width = data[0]['video_height'], data[0]['video_width'] + video_length = len(data) + + second_person_base_colors_255_dict = { + # Warm Colors for Right Side (R.) - Red, Orange, Yellow + "Red": [255, 20, 20], + "Orange": [255, 60, 0], + "Golden Orange": [255, 110, 0], + "Yellow": [255, 200, 0], + "Yellow-Green": [160, 255, 40], + + # Cool Colors for Left Side (L.) - Green, Blue, Purple + "Bright Green": [0, 255, 50], + "Light Green-Blue": [0, 255, 100], + "Aqua": [0, 255, 200], + "Cyan": [0, 230, 255], + "Sky Blue": [0, 130, 255], + "Medium Blue": [0, 70, 255], + "Pure Blue": [0, 0, 255], + "Purple-Blue": [80, 0, 255], + "Medium Purple": [160, 0, 255], + + # Neutral/Central Colors (e.g., for Neck, Nose, Eyes, Ears) + "Grey": [130, 130, 130], + "Pink-Magenta": [255, 0, 150], + "Dark Pink": [255, 0, 100], + "Violet": [120, 0, 255], + "Dark Violet": [60, 0, 255], + } + + first_person_base_colors_255_dict = { + # Warm Colors for Right Side (R.) - Red, Orange, Yellow + "Red": [255, 150, 150], + "Orange": [255, 180, 140], + "Golden Orange": [255, 215, 150], + "Yellow": [255, 240, 170], + "Yellow-Green": [200, 255, 100], + + # Cool Colors for Left Side (L.) - Green, Blue, Purple + "Bright Green": [100, 255, 100], + "Light Green-Blue": [140, 255, 180], + "Aqua": [150, 240, 200], + "Cyan": [180, 230, 240], + "Sky Blue": [160, 200, 255], + "Medium Blue": [100, 120, 255], + "Pure Blue": [120, 140, 255], + "Purple-Blue": [180, 90, 255], + "Medium Purple": [190, 120, 255], + + # Neutral/Central Colors (e.g., for Neck, Nose, Eyes, Ears) + "Grey": [210, 210, 210], + "Pink-Magenta": [255, 120, 200], + "Dark Pink": [255, 150, 180], + "Violet": [200, 90, 255], + "Dark Violet": [130, 80, 255], + } + + base_colors_255_dict_list = [first_person_base_colors_255_dict, second_person_base_colors_255_dict] + ordered_colors_255_list = [[ + base_colors_255_dict["Red"], # Neck -> R. Shoulder (Red) + base_colors_255_dict["Cyan"], # Neck -> L. Shoulder (Cyan) + base_colors_255_dict["Orange"], # R. Shoulder -> R. Elbow (Orange) + base_colors_255_dict["Golden Orange"], # R. Elbow -> R. Wrist (Golden Orange) + base_colors_255_dict["Sky Blue"], # L. Shoulder -> L. Elbow (Sky Blue) + base_colors_255_dict["Medium Blue"], # L. Elbow -> L. Wrist (Medium Blue) + base_colors_255_dict["Yellow-Green"], # Neck -> R. Hip ( Yellow-Green) + base_colors_255_dict["Bright Green"], # R. Hip -> R. Knee (Bright Green - transitioning warm to cool spectrum) + base_colors_255_dict["Light Green-Blue"], # R. Knee -> R. Ankle (Light Green-Blue - transitioning) + base_colors_255_dict["Pure Blue"], # Neck -> L. Hip (Pure Blue) + base_colors_255_dict["Purple-Blue"], # L. Hip -> L. Knee (Purple-Blue) + base_colors_255_dict["Medium Purple"], # L. Knee -> L. Ankle (Medium Purple) + base_colors_255_dict["Grey"], # Neck -> Nose (Grey) + base_colors_255_dict["Pink-Magenta"], # Nose -> R. Eye (Pink/Magenta) + base_colors_255_dict["Dark Violet"], # R. Eye -> R. Ear (Dark Pink) + base_colors_255_dict["Pink-Magenta"], # Nose -> L. Eye (Violet) + base_colors_255_dict["Dark Violet"], # L. Eye -> L. Ear (Dark Violet) + ] for base_colors_255_dict in base_colors_255_dict_list] + + limb_seq = [ + [1, 2], # 0 Neck -> R. Shoulder + [1, 5], # 1 Neck -> L. Shoulder + [2, 3], # 2 R. Shoulder -> R. Elbow + [3, 4], # 3 R. Elbow -> R. Wrist + [5, 6], # 4 L. Shoulder -> L. Elbow + [6, 7], # 5 L. Elbow -> L. Wrist + [1, 8], # 6 Neck -> R. Hip + [8, 9], # 7 R. Hip -> R. Knee + [9, 10], # 8 R. Knee -> R. Ankle + [1, 11], # 9 Neck -> L. Hip + [11, 12], # 10 L. Hip -> L. Knee + [12, 13], # 11 L. Knee -> L. Ankle + [1, 0], # 12 Neck -> Nose + [0, 14], # 13 Nose -> R. Eye + [14, 16], # 14 R. Eye -> R. Ear + [0, 15], # 15 Nose -> L. Eye + [15, 17], # 16 L. Eye -> L. Ear + ] + + draw_seq = [0, 2, 3, # Neck -> R. Shoulder -> R. Elbow -> R. Wrist + 1, 4, 5, # Neck -> L. Shoulder -> L. Elbow -> L. Wrist + 6, 7, 8, # Neck -> R. Hip -> R. Knee -> R. Ankle + 9, 10, 11, # Neck -> L. Hip -> L. Knee -> L. Ankle + 12, # Neck -> Nose + 13, 14, # Nose -> R. Eye -> R. Ear + 15, 16, # Nose -> L. Eye -> L. Ear + ] # Expanding outward from the proximal end + + colors_first = [[c / 300 + 0.15 for c in color_rgb] + [0.8] for color_rgb in ordered_colors_255_list[0]] + colors_second = [[c / 300 + 0.15 for c in color_rgb] + [0.8] for color_rgb in ordered_colors_255_list[1]] + + smpl_poses_first, smpl_poses_second = collect_smpl_poses_samurai(data) + + + if intrinsic_matrix is None: + intrinsic_matrix = intrinsic_matrix_from_field_of_view((height, width)) + focal_x = intrinsic_matrix[0,0] + focal_y = intrinsic_matrix[1,1] + princpt = (intrinsic_matrix[0,2], intrinsic_matrix[1,2]) # (cx, cy) + + # obtain cylinder_specs for each frame + cylinder_specs_list = [] + for i in range(video_length): + cylinder_specs_first = get_single_pose_cylinder_specs((i, smpl_poses_first[i], None, None, None, None, colors_first, limb_seq, draw_seq)) + cylinder_specs_second = get_single_pose_cylinder_specs((i, smpl_poses_second[i], None, None, None, None, colors_second, limb_seq, draw_seq)) + cylinder_specs = cylinder_specs_first + cylinder_specs_second + cylinder_specs_list.append(cylinder_specs) + + + frames_np_rgba = render_whole(cylinder_specs_list, H=height, W=width, fx=focal_x, fy=focal_y, cx=princpt[0], cy=princpt[1]) + if dw_poses is not None and draw_2d: + aligned_poses = copy.deepcopy(dw_poses) + canvas_2d = draw_pose_to_canvas_np(aligned_poses, pool=None, H=height, W=width, reshape_scale=0, show_feet_flag=False, show_body_flag=False, show_cheek_flag=True, dw_hand=True) + for i in range(len(frames_np_rgba)): + frame_img = frames_np_rgba[i] + canvas_img = canvas_2d[i] + mask = canvas_img != 0 + frame_img[:, :, :3][mask] = canvas_img[mask] + frames_np_rgba[i] = frame_img + + return frames_np_rgba diff --git a/__init__.py b/__init__.py new file mode 100644 index 0000000..2e96bd6 --- /dev/null +++ b/__init__.py @@ -0,0 +1,3 @@ +from .nodes import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS + +__all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS"] \ No newline at end of file diff --git a/example_workflows/SCAIL_preprocess_example_01.json b/example_workflows/SCAIL_preprocess_example_01.json new file mode 100644 index 0000000..0b9b4c4 --- /dev/null +++ b/example_workflows/SCAIL_preprocess_example_01.json @@ -0,0 +1,825 @@ +{ + "id": "c6e410bc-5e2c-460b-ae81-c91b6094fbb1", + "revision": 0, + "last_node_id": 381, + "last_link_id": 706, + "nodes": [ + { + "id": 379, + "type": "VHS_LoadVideo", + "pos": [ + -728.1399554193044, + -1903.2296105502905 + ], + "size": [ + 255.8291015625, + 743.5855352640658 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [ + { + "name": "meta_batch", + "shape": 7, + "type": "VHS_BatchManager", + "link": null + }, + { + "name": "vae", + "shape": 7, + "type": "VAE", + "link": null + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 684 + ] + }, + { + "name": "frame_count", + "type": "INT", + "links": [] + }, + { + "name": "audio", + "type": "AUDIO", + "links": null + }, + { + "name": "video_info", + "type": "VHS_VIDEOINFO", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "8550981384301e9bc5bfea83e5c2c75258102593", + "Node name for S&R": "VHS_LoadVideo" + }, + "widgets_values": { + "video": "vid.mp4", + "force_rate": 0, + "custom_width": 0, + "custom_height": 0, + "frame_load_cap": 81, + "skip_first_frames": 0, + "select_every_nth": 2, + "format": "AnimateDiff", + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "vid.mp4", + "type": "input", + "format": "video/mp4", + "force_rate": 0, + "custom_width": 0, + "custom_height": 0, + "frame_load_cap": 81, + "skip_first_frames": 0, + "select_every_nth": 2 + } + } + } + }, + { + "id": 358, + "type": "OnnxDetectionModelLoader", + "pos": [ + -731.2840742077292, + -2127.3788215536956 + ], + "size": [ + 304.0484375, + 106 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "model", + "type": "POSEMODEL", + "links": [ + 677, + 680 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanAnimatePreprocess", + "ver": "65502e208ae89231619e3def41bc8bafe6fc4f1e", + "Node name for S&R": "OnnxDetectionModelLoader" + }, + "widgets_values": [ + "vitpose-l-wholebody.onnx", + "onnx\\yolov10m.onnx", + "CUDAExecutionProvider" + ] + }, + { + "id": 361, + "type": "NLFPredict", + "pos": [ + 485.2315622592827, + -2157.1952980266337 + ], + "size": [ + 157.564453125, + 46 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "NLFMODEL", + "link": 627 + }, + { + "name": "images", + "type": "IMAGE", + "link": 701 + } + ], + "outputs": [ + { + "name": "pose_results", + "type": "NLFPRED", + "links": [ + 656 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "056d8ad96ffa5a223a8cd88c900a573a8d450e22", + "Node name for S&R": "NLFPredict" + }, + "widgets_values": [] + }, + { + "id": 362, + "type": "DownloadAndLoadNLFModel", + "pos": [ + -279.5657072597091, + -2231.190892106357 + ], + "size": [ + 606.9488335754912, + 82 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "nlf_model", + "type": "NLFMODEL", + "links": [ + 627 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "056d8ad96ffa5a223a8cd88c900a573a8d450e22", + "Node name for S&R": "DownloadAndLoadNLFModel" + }, + "widgets_values": [ + "https://github.com/isarandi/nlf/releases/download/v0.3.2/nlf_l_multi_0.3.2.torchscript", + true + ] + }, + { + "id": 359, + "type": "LoadImage", + "pos": [ + -369.63293579008285, + -1510.3591727642465 + ], + "size": [ + 282.798828125, + 314 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 698 + ] + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.4.0", + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "anya.jpg", + "image" + ] + }, + { + "id": 381, + "type": "Reroute", + "pos": [ + 144.49613132898912, + -2052.387551799464 + ], + "size": [ + 75, + 26 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "name": "", + "type": "*", + "link": 700 + } + ], + "outputs": [ + { + "name": "", + "type": "IMAGE", + "links": [ + 701, + 702 + ] + } + ], + "properties": { + "showOutputText": false, + "horizontal": false + } + }, + { + "id": 380, + "type": "VHS_VideoCombine", + "pos": [ + 1061.113369865757, + -2147.425691347535 + ], + "size": [ + 329.65692831178876, + 889.8996245456303 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 685 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": null + }, + { + "name": "meta_batch", + "shape": 7, + "type": "VHS_BatchManager", + "link": null + }, + { + "name": "vae", + "shape": 7, + "type": "VAE", + "link": null + } + ], + "outputs": [ + { + "name": "Filenames", + "type": "VHS_FILENAMES", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "8550981384301e9bc5bfea83e5c2c75258102593", + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 16, + "loop_count": 0, + "filename_prefix": "SCAIL_pose", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 19, + "save_metadata": true, + "trim_to_audio": false, + "pingpong": false, + "save_output": false, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "SCAIL_pose_00030.mp4", + "subfolder": "", + "type": "temp", + "format": "video/h264-mp4", + "frame_rate": 16, + "workflow": "SCAIL_pose_00030.png", + "fullpath": "N:\\AI\\ComfyUI\\temp\\SCAIL_pose_00030.mp4" + } + } + } + }, + { + "id": 376, + "type": "PoseDetectionVitPoseToDWPose", + "pos": [ + 330.07991567045417, + -1961.6148428238132 + ], + "size": [ + 271.9595703125, + 46 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "vitpose_model", + "type": "POSEMODEL", + "link": 677 + }, + { + "name": "images", + "type": "IMAGE", + "link": 702 + } + ], + "outputs": [ + { + "name": "dw_poses", + "type": "DWPOSES", + "links": [ + 704 + ] + } + ], + "properties": { + "Node name for S&R": "PoseDetectionVitPoseToDWPose" + }, + "widgets_values": [] + }, + { + "id": 370, + "type": "RenderNLFPoses", + "pos": [ + 747.8706167185883, + -2157.040108369948 + ], + "size": [ + 270, + 146 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "nlf_poses", + "type": "NLFPRED", + "link": 656 + }, + { + "name": "dw_poses", + "shape": 7, + "type": "DWPOSES", + "link": 704 + }, + { + "name": "ref_dw_pose", + "shape": 7, + "type": "DWPOSES", + "link": 706 + }, + { + "name": "width", + "type": "INT", + "widget": { + "name": "width" + }, + "link": 688 + }, + { + "name": "height", + "type": "INT", + "widget": { + "name": "height" + }, + "link": 689 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 685 + ] + }, + { + "name": "mask", + "type": "MASK", + "links": [] + } + ], + "properties": { + "Node name for S&R": "RenderNLFPoses" + }, + "widgets_values": [ + 512, + 896 + ] + }, + { + "id": 372, + "type": "ImageResizeKJv2", + "pos": [ + -379.6838809890405, + -1915.7782190865344 + ], + "size": [ + 270, + 336 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 684 + }, + { + "name": "mask", + "shape": 7, + "type": "MASK", + "link": null + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 700 + ] + }, + { + "name": "width", + "type": "INT", + "links": [ + 688, + 695 + ] + }, + { + "name": "height", + "type": "INT", + "links": [ + 689, + 696 + ] + }, + { + "name": "mask", + "type": "MASK", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "1b9af2322c009e0171f1c7c2f8457ca69afa1ec0", + "Node name for S&R": "ImageResizeKJv2" + }, + "widgets_values": [ + 512, + 896, + "bilinear", + "crop", + "0, 0, 0", + "center", + 2, + "cpu", + "Output: 81 x 512 x 896 | 425.25MB" + ] + }, + { + "id": 374, + "type": "ImageResizeKJv2", + "pos": [ + -46.738037813889264, + -1635.669074975477 + ], + "size": [ + 270, + 336 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 698 + }, + { + "name": "mask", + "shape": 7, + "type": "MASK", + "link": null + }, + { + "name": "width", + "type": "INT", + "widget": { + "name": "width" + }, + "link": 695 + }, + { + "name": "height", + "type": "INT", + "widget": { + "name": "height" + }, + "link": 696 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 682 + ] + }, + { + "name": "width", + "type": "INT", + "links": null + }, + { + "name": "height", + "type": "INT", + "links": null + }, + { + "name": "mask", + "type": "MASK", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "1b9af2322c009e0171f1c7c2f8457ca69afa1ec0", + "Node name for S&R": "ImageResizeKJv2" + }, + "widgets_values": [ + 512, + 896, + "bilinear", + "pad", + "0, 0, 0", + "center", + 2, + "cpu", + "Output: 1 x 512 x 896 | 5.25MB" + ] + }, + { + "id": 377, + "type": "PoseDetectionVitPoseToDWPose", + "pos": [ + 361.20948481109036, + -1818.3087110204879 + ], + "size": [ + 271.9595703125, + 46 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "vitpose_model", + "type": "POSEMODEL", + "link": 680 + }, + { + "name": "images", + "type": "IMAGE", + "link": 682 + } + ], + "outputs": [ + { + "name": "dw_poses", + "type": "DWPOSES", + "links": [ + 706 + ] + } + ], + "properties": { + "Node name for S&R": "PoseDetectionVitPoseToDWPose" + }, + "widgets_values": [] + } + ], + "links": [ + [ + 627, + 362, + 0, + 361, + 0, + "NLFMODEL" + ], + [ + 656, + 361, + 0, + 370, + 0, + "NLFPRED" + ], + [ + 677, + 358, + 0, + 376, + 0, + "POSEMODEL" + ], + [ + 680, + 358, + 0, + 377, + 0, + "POSEMODEL" + ], + [ + 682, + 374, + 0, + 377, + 1, + "IMAGE" + ], + [ + 684, + 379, + 0, + 372, + 0, + "IMAGE" + ], + [ + 685, + 370, + 0, + 380, + 0, + "IMAGE" + ], + [ + 688, + 372, + 1, + 370, + 3, + "INT" + ], + [ + 689, + 372, + 2, + 370, + 4, + "INT" + ], + [ + 695, + 372, + 1, + 374, + 2, + "INT" + ], + [ + 696, + 372, + 2, + 374, + 3, + "INT" + ], + [ + 698, + 359, + 0, + 374, + 0, + "IMAGE" + ], + [ + 700, + 372, + 0, + 381, + 0, + "IMAGE" + ], + [ + 701, + 381, + 0, + 361, + 1, + "IMAGE" + ], + [ + 702, + 381, + 0, + 376, + 1, + "IMAGE" + ], + [ + 704, + 376, + 0, + 370, + 1, + "DWPOSES" + ], + [ + 706, + 377, + 0, + 370, + 2, + "DWPOSES" + ] + ], + "groups": [], + "config": {}, + "extra": { + "ds": { + "scale": 0.8140274938684753, + "offset": [ + 1341.9401489123059, + 2475.387199967974 + ] + }, + "frontendVersion": "1.35.3", + "workflowRendererVersion": "LG", + "node_versions": { + "ComfyUI-WanVideoWrapper": "5a2383621a05825d0d0437781afcb8552d9590fd", + "comfy-core": "0.3.26", + "ComfyUI-VideoHelperSuite": "0a75c7958fe320efcb052f1d9f8451fd20c730a8" + }, + "VHS_latentpreview": true, + "VHS_latentpreviewrate": 0, + "VHS_MetadataImage": true, + "VHS_KeepIntermediate": true + }, + "version": 0.4 +} \ No newline at end of file diff --git a/nodes.py b/nodes.py new file mode 100644 index 0000000..a8b2078 --- /dev/null +++ b/nodes.py @@ -0,0 +1,243 @@ +import os +import torch +from tqdm import tqdm +import numpy as np +import folder_paths +import cv2 +import logging +import copy +script_directory = os.path.dirname(os.path.abspath(__file__)) + +from comfy import model_management as mm +from comfy.utils import ProgressBar +device = mm.get_torch_device() +offload_device = mm.unet_offload_device() + +folder_paths.add_model_folder_path("detection", os.path.join(folder_paths.models_dir, "detection")) + +from .vitpose_utils.utils import bbox_from_detector, crop, load_pose_metas_from_kp2ds_seq, aaposemeta_to_dwpose_scail + +def scale_faces(poses, pose_2d_ref): + # Input: two lists of dict, poses[0]['faces'].shape: 1, 68, 2 , poses_ref[0]['faces'].shape: 1, 68, 2 + # Scale the facial keypoints in poses according to the center point of the face + # That is: calculate the distance from the center point (idx: 30) to other facial keypoints in ref, + # and the same for poses, then get scale_n as the ratio + # Clamp scale_n to the range 0.8-1.5, then apply it to poses + # Note: poses are modified in place + + ref = pose_2d_ref[0] + pose_0 = poses[0] + + face_0 = pose_0['faces'] # shape: (1, 68, 2) + face_ref = ref['faces'] + + # Extract numpy arrays + face_0 = np.array(face_0[0]) # (68, 2) + face_ref = np.array(face_ref[0]) + + # Center point (nose tip or face center) + center_idx = 30 + center_0 = face_0[center_idx] + center_ref = face_ref[center_idx] + + # Calculate distance to center point + dist = np.linalg.norm(face_0 - center_0, axis=1) + dist_ref = np.linalg.norm(face_ref - center_ref, axis=1) + + # Avoid the 0 distance of the center point itself + dist = np.delete(dist, center_idx) + dist_ref = np.delete(dist_ref, center_idx) + + mean_dist = np.mean(dist) + mean_dist_ref = np.mean(dist_ref) + + if mean_dist < 1e-6: + scale_n = 1.0 + else: + scale_n = mean_dist_ref / mean_dist + + # Clamp to [0.8, 1.5] + scale_n = np.clip(scale_n, 0.8, 1.5) + + for i, pose in enumerate(poses): + face = pose['faces'] + # Extract numpy array + face = np.array(face[0]) # (68, 2) + center = face[center_idx] + scaled_face = (face - center) * scale_n + center + poses[i]['faces'][0] = scaled_face + + body = pose['bodies'] + candidate = body['candidate'] + candidate_np = np.array(candidate[0]) # (14, 2) + body_center = candidate_np[0] + scaled_candidate = (candidate_np - body_center) * scale_n + body_center + poses[i]['bodies']['candidate'][0] = scaled_candidate + + # In-place modification + pose['faces'][0] = scaled_face + + return scale_n + +class PoseDetectionVitPoseToDWPose: + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "vitpose_model": ("POSEMODEL",), + "images": ("IMAGE",), + }, + } + + RETURN_TYPES = ("DWPOSES",) + RETURN_NAMES = ("dw_poses",) + FUNCTION = "process" + CATEGORY = "WanAnimatePreprocess" + DESCRIPTION = "ViTPose to DWPose format pose detection node." + + def process(self, vitpose_model, images): + + detector = vitpose_model["yolo"] + pose_model = vitpose_model["vitpose"] + B, H, W, C = images.shape + + shape = np.array([H, W])[None] + images_np = images.numpy() + + IMG_NORM_MEAN = np.array([0.485, 0.456, 0.406]) + IMG_NORM_STD = np.array([0.229, 0.224, 0.225]) + input_resolution=(256, 192) + rescale = 1.25 + + detector.reinit() + pose_model.reinit() + + comfy_pbar = ProgressBar(B*2) + progress = 0 + bboxes = [] + for img in tqdm(images_np, total=len(images_np), desc="Detecting bboxes"): + bboxes.append(detector( + cv2.resize(img, (640, 640)).transpose(2, 0, 1)[None], + shape + )[0][0]["bbox"]) + progress += 1 + if progress % 10 == 0: + comfy_pbar.update_absolute(progress) + + detector.cleanup() + + kp2ds = [] + for img, bbox in tqdm(zip(images_np, bboxes), total=len(images_np), desc="Extracting keypoints"): + if bbox is None or bbox[-1] <= 0 or (bbox[2] - bbox[0]) < 10 or (bbox[3] - bbox[1]) < 10: + bbox = np.array([0, 0, img.shape[1], img.shape[0]]) + + bbox_xywh = bbox + center, scale = bbox_from_detector(bbox_xywh, input_resolution, rescale=rescale) + img = crop(img, center, scale, (input_resolution[0], input_resolution[1]))[0] + + img_norm = (img - IMG_NORM_MEAN) / IMG_NORM_STD + img_norm = img_norm.transpose(2, 0, 1).astype(np.float32) + + keypoints = pose_model(img_norm[None], np.array(center)[None], np.array(scale)[None]) + kp2ds.append(keypoints) + progress += 1 + if progress % 10 == 0: + comfy_pbar.update_absolute(progress) + + pose_model.cleanup() + + kp2ds = np.concatenate(kp2ds, 0) + pose_metas = load_pose_metas_from_kp2ds_seq(kp2ds, width=W, height=H) + dwposes = [aaposemeta_to_dwpose_scail(meta) for meta in pose_metas] + + return (dwposes,) + + +class RenderNLFPoses: + @classmethod + def INPUT_TYPES(s): + return {"required": { + "nlf_poses": ("NLFPRED", {"tooltip": "Input poses for the model"}), + "width": ("INT", {"default": 512}), + "height": ("INT", {"default": 512}), + }, + "optional": { + "dw_poses": ("DWPOSES", {"default": None, "tooltip": "Optional DW pose model for 2D drawing"}), + "ref_dw_pose": ("DWPOSES", {"default": None, "tooltip": "Optional reference DW pose model for alignment"}), + } + } + + RETURN_TYPES = ("IMAGE", "MASK",) + RETURN_NAMES = ("image", "mask",) + FUNCTION = "predict" + CATEGORY = "WanVideoWrapper" + + def predict(self, nlf_poses, width, height, dw_poses=None, ref_dw_pose=None): + + from .NLFPoseExtract.nlf_render import render_nlf_as_images, shift_dwpose_according_to_nlf, process_data_to_COCO_format, intrinsic_matrix_from_field_of_view + from .NLFPoseExtract.align3d import solve_new_camera_params_central, solve_new_camera_params_down + + if isinstance(nlf_poses, dict): + pose_input = nlf_poses['joints3d_nonparam'][0] if 'joints3d_nonparam' in nlf_poses else nlf_poses + else: + pose_input = nlf_poses + + dw_pose_input = copy.deepcopy(dw_poses) + + ori_camera_pose = intrinsic_matrix_from_field_of_view([height, width]) + ori_focal = ori_camera_pose[0, 0] + + if ref_dw_pose is not None: + ref_dw_pose_input = copy.deepcopy(ref_dw_pose) + pose_3d_first_driving_frame = pose_input[0][0].cpu().numpy() + pose_3d_coco_first_driving_frame = process_data_to_COCO_format(pose_3d_first_driving_frame) + poses_2d_ref = ref_dw_pose_input[0]['bodies']['candidate'][0][:14] + poses_2d_ref[:, 0] = poses_2d_ref[:, 0] * width + poses_2d_ref[:, 1] = poses_2d_ref[:, 1] * height + + poses_2d_subset = ref_dw_pose[0]['bodies']['subset'][0][:14] + pose_3d_coco_first_driving_frame = pose_3d_coco_first_driving_frame[:14] + + valid_indices, valid_upper_indices, valid_lower_indices = [], [], [] + upper_body_indices = [0, 2, 3, 5, 6] + lower_body_indices = [9, 10, 12, 13] + + for i in range(len(poses_2d_subset)): + if poses_2d_subset[i] != -1.0 and np.sum(pose_3d_coco_first_driving_frame[i]) != 0: + if i in upper_body_indices: + valid_upper_indices.append(i) + if i in lower_body_indices: + valid_lower_indices.append(i) + + valid_indices = [1] + valid_lower_indices if len(valid_upper_indices) < 4 else [1] + valid_lower_indices + valid_upper_indices # align body or only lower body + + pose_2d_ref = poses_2d_ref[valid_indices] + pose_3d_coco_first_driving_frame = pose_3d_coco_first_driving_frame[valid_indices] + + if len(valid_lower_indices) >= 4: + new_camera_intrinsics, scale_m = solve_new_camera_params_down(pose_3d_coco_first_driving_frame, ori_focal, [height, width], pose_2d_ref) + else: + new_camera_intrinsics, scale_m = solve_new_camera_params_central(pose_3d_coco_first_driving_frame, ori_focal, [height, width], pose_2d_ref) + + scale_face = scale_faces(list(dw_pose_input), list(ref_dw_pose)) # poses[0]['faces'].shape: 1, 68, 2 , poses_ref[0]['faces'].shape: 1, 68, 2 + + logging.info(f"Scale - m: {scale_m}, face: {scale_face}") + shift_dwpose_according_to_nlf(pose_input, dw_pose_input, ori_camera_pose, new_camera_intrinsics, height, width) + + frames_np = render_nlf_as_images(pose_input, dw_pose_input, height, width, len(pose_input), intrinsic_matrix=new_camera_intrinsics) + else: + frames_np = render_nlf_as_images(pose_input, dw_pose_input, height, width, len(pose_input), intrinsic_matrix=ori_camera_pose) + + frames_tensor = torch.from_numpy(np.stack(frames_np, axis=0)).contiguous() / 255.0 + frames_tensor, mask = frames_tensor[..., :3], frames_tensor[..., -1] > 0.5 + + return (frames_tensor.cpu().float(), mask.cpu().float()) + +NODE_CLASS_MAPPINGS = { + "PoseDetectionVitPoseToDWPose": PoseDetectionVitPoseToDWPose, + "RenderNLFPoses": RenderNLFPoses, +} +NODE_DISPLAY_NAME_MAPPINGS = { + "PoseDetectionVitPoseToDWPose": "Pose Detection VitPose to DWPose", + "RenderNLFPoses": "Render NLF Poses", +} diff --git a/pose_draw/draw_3d_utils.py b/pose_draw/draw_3d_utils.py new file mode 100644 index 0000000..5af2725 --- /dev/null +++ b/pose_draw/draw_3d_utils.py @@ -0,0 +1,221 @@ +import numpy as np + +def convert_3dpose_to_2dpose_body(body_keypoints, face_keypoints): + """ + Map 20-point 3D coordinates to 18-point 2D coordinates. + :param poses: Input list of 20 coordinates, each point as [x, y, z] + :return: Mapped list of 18 coordinates, each point as [x, y] + """ + # Mapping relationship: index positions + body_mapping = { + 0: 1, 1: 2, 2: 3, 3: 4, 4: 5, 5: 6, 6: 7, 7: 8, 8: 8, 9: 9, 10: 10, 11: 23, 13: 22, 12: 21, + 14: 11, 15: 12, 16: 13, 17: 20, 18: 18, 19: 19 + } + face_mapping = { + 1: 16, 8: 14, 4: 0, 7: 15, 0: 17 + } + + # Initialize 18-point coordinate list, default value is [-1, -1] + result = [[-1, -1] for _ in range(24)] + + # Traverse the mapping relationship, map the corresponding 20-point coordinates to 18-point coordinates + for src_idx, dst_idx in body_mapping.items(): + if src_idx < len(body_keypoints): # 确保索引不越界 + result[dst_idx] = [body_keypoints[src_idx][1],body_keypoints[src_idx][0]] # Extract x, y coordinates + for src_idx, dst_idx in face_mapping.items(): + if src_idx < len(face_keypoints): + result[dst_idx] = [face_keypoints[src_idx][1], face_keypoints[src_idx][0]] + return result + +def convert_3dpose_to_2dpose_hand(left_hand_keypoints, right_hand_keypoints, body_keypoints): + """ + Map 20-point 3D coordinates to 18-point 2D coordinates. + :param poses: Input list of 20 coordinates, each point as [x, y, z] + :return: Mapped list of 18 coordinates, each point as [x, y] + """ + # Mapping relationship: index positions + hand_mapping = { + 0: 1, 1: 2, 2: 3, 3: 4, 4: 5, 5: 6, 6: 7, 7: 8, 8: 9, 9: 10, + 10: 11, 11: 12, 12: 13, 13: 14, 14: 15, 15: 16, 16: 17, 17: 18, + 18: 19, 19: 20 + } + + body_mapping_left = {3: 0} + body_mapping_right = {6: 0} + + # Initialize 18-point coordinate list, default value is [-1, -1] + left_result = [[-1, -1] for _ in range(21)] + right_result = [[-1, -1] for _ in range(21)] + + # Traverse the mapping relationship, map the corresponding 20-point coordinates to 18-point coordinates + for src_idx, dst_idx in hand_mapping.items(): + if src_idx < len(left_hand_keypoints): # 确保索引不越界 + left_result[dst_idx] = [left_hand_keypoints[src_idx][1], left_hand_keypoints[src_idx][0]] # Extract x, y coordinates + right_result[dst_idx] = [right_hand_keypoints[src_idx][1], right_hand_keypoints[src_idx][0]] + + for src_idx, dst_idx in body_mapping_left.items(): + if src_idx < len(body_keypoints): + left_result[dst_idx] = [body_keypoints[src_idx][1], body_keypoints[src_idx][0]] + for src_idx, dst_idx in body_mapping_right.items(): + if src_idx < len(body_keypoints): + right_result[dst_idx] = [body_keypoints[src_idx][1], body_keypoints[src_idx][0]] + + return [left_result, right_result] + +def convert_3dpose_to_2dpose_face(face_keypoints): + # Set [-1, -1] for indices 0, 1, 4, 5, 6, 7, 8, otherwise extract [y, x] + result = [[-1, -1] if i in [0, 1, 4, 5, 6, 7, 8] else [pt[1], pt[0]] for i, pt in enumerate(face_keypoints)] + return result + +def correct_lift_end_kpt_by_phmr(start, end, dwpose_kpts, lift_start, lift_end, phmr_start, phmr_end): + ''' + Check if the other end meets the requirements. If so, return the result after lift, otherwise return the phmr result. + ''' + if dwpose_kpts[start][0] == -1: + return + lift_vec = np.array(lift_end) - np.array(lift_start) + phmr_vec = np.array(phmr_end) - np.array(phmr_start) + start_distance = np.linalg.norm(np.array(lift_start) - np.array(phmr_start)) + end_distance = np.linalg.norm(np.array(lift_end) - np.array(phmr_end)) + lift_vec_len = np.linalg.norm(lift_vec) + phmr_vec_len = np.linalg.norm(phmr_vec) + if start_distance + end_distance > phmr_vec_len: + dwpose_kpts[end] = [-1, -1] + theta = np.arccos(np.dot(lift_vec, phmr_vec) / (lift_vec_len * phmr_vec_len)) + if lift_vec_len > phmr_vec_len * 1.65 or lift_vec_len < phmr_vec_len * 0.4 or theta > np.pi / 4: + dwpose_kpts[end] = [-1, -1] + return + + + +def mix_3d_poses(poses_dwpose, poses_3dpose): + ''' + Combine two types of poses: use the body from 3dPose, and the face and hand from DWPose. + ''' + poses = [] + for pose_dwpose, pose_3dpose in zip(poses_dwpose, poses_3dpose): + pose = { + "bodies": { + "candidate": pose_3dpose["bodies"]["candidate"], + "subset": pose_dwpose["bodies"]["subset"] + }, + "faces": pose_dwpose["faces"], + "hands": pose_dwpose["hands"] + } + poses.append(pose) + return poses + +def correct_hand_from_3d(hand_keypoints_dwpose, hand_keypoints_3dpose): + ''' + If the hand keypoints of dwpose and 3dpose differ too much, remove the farthest end. + ''' + edges_palm = [ + [1, 2], [2, 3], [3, 4], + [5, 6], [6, 7], [7, 8], + [9, 10], [10, 11], [11, 12], + [13, 14], [14, 15], [15, 16], + [17, 18], [18, 19], [19, 20], + ] + edges_finger = [[0, 1], [0, 5], [0, 9], [0, 13], [0, 17]] + max_length_palm = 0 + max_length_finger = 0 + for edge in edges_palm: + limb_length_3dpose = np.linalg.norm(np.array(hand_keypoints_3dpose[edge[0]]) - np.array(hand_keypoints_3dpose[edge[1]])) + if limb_length_3dpose > max_length_palm: + max_length_palm = limb_length_3dpose + for edge in edges_finger: + limb_length_3dpose = np.linalg.norm(np.array(hand_keypoints_3dpose[edge[0]]) - np.array(hand_keypoints_3dpose[edge[1]])) + if limb_length_3dpose > max_length_finger: + max_length_finger = limb_length_3dpose + for edge in edges_palm: + limb_length_dwpose = np.linalg.norm(np.array(hand_keypoints_dwpose[edge[0]]) - np.array(hand_keypoints_dwpose[edge[1]])) + if limb_length_dwpose > max_length_palm * 1.5: + if -1 in hand_keypoints_dwpose[edge[0]] or -1 in hand_keypoints_dwpose[edge[1]] or -1 in hand_keypoints_3dpose[edge[0]] or -1 in hand_keypoints_3dpose[edge[1]]: + continue + distance_point_0 = np.linalg.norm(np.array(hand_keypoints_dwpose[edge[0]]) - np.array(hand_keypoints_3dpose[edge[0]])) + distance_point_1 = np.linalg.norm(np.array(hand_keypoints_dwpose[edge[1]]) - np.array(hand_keypoints_3dpose[edge[1]])) + if distance_point_0 > distance_point_1: + hand_keypoints_dwpose[edge[1]] = [-1, -1] + else: + hand_keypoints_dwpose[edge[0]] = [-1, -1] + for edge in edges_finger: + limb_length_dwpose = np.linalg.norm(np.array(hand_keypoints_dwpose[edge[0]]) - np.array(hand_keypoints_dwpose[edge[1]])) + if limb_length_dwpose > max_length_finger * 1.5: + if -1 in hand_keypoints_dwpose[edge[0]] or -1 in hand_keypoints_dwpose[edge[1]] or -1 in hand_keypoints_3dpose[edge[0]] or -1 in hand_keypoints_3dpose[edge[1]]: + continue + distance_point_0 = np.linalg.norm(np.array(hand_keypoints_dwpose[edge[0]]) - np.array(hand_keypoints_3dpose[edge[0]])) + distance_point_1 = np.linalg.norm(np.array(hand_keypoints_dwpose[edge[1]]) - np.array(hand_keypoints_3dpose[edge[1]])) + if distance_point_0 > distance_point_1: + hand_keypoints_dwpose[edge[1]] = [-1, -1] + else: + hand_keypoints_dwpose[edge[0]] = [-1, -1] + return hand_keypoints_dwpose + +def correct_body_from_3d(body_keypoints_dwpose, body_keypoints_3dpose, subset_dwpose, subset_3dpose): + ''' + If the bone length of dwpose and 3dpose differ too much, remove the farthest end. + ''' + limbSeq = [ + [2, 3], + [2, 6], + [3, 4], + [4, 5], + [6, 7], + [7, 8], + [2, 9], + [9, 10], + [10, 11], + [2, 12], + [12, 13], + [13, 14], + [2, 1], + [1, 15], + [15, 17], + [1, 16], + [16, 18], + [3, 17], + [6, 18], + ] + + for ori_limb in limbSeq: + limb = [ori_limb[0] - 1, ori_limb[1] - 1] + limb_length_dwpose = np.linalg.norm(np.array(body_keypoints_dwpose[limb[0]]) - np.array(body_keypoints_dwpose[limb[1]])) + limb_length_3dpose = np.linalg.norm(np.array(body_keypoints_3dpose[limb[0]]) - np.array(body_keypoints_3dpose[limb[1]])) + if subset_dwpose[0][limb[0]] == -1 or subset_dwpose[0][limb[1]] == -1 or subset_3dpose[0][limb[0]] == -1 or subset_3dpose[0][limb[1]] == -1: + continue + if limb_length_dwpose > limb_length_3dpose * 2: + # Determine the farther end + distance_point_0 = np.linalg.norm(np.array(body_keypoints_dwpose[limb[0]]) - np.array(body_keypoints_3dpose[limb[0]])) + distance_point_1 = np.linalg.norm(np.array(body_keypoints_dwpose[limb[1]]) - np.array(body_keypoints_3dpose[limb[1]])) + if distance_point_0 > distance_point_1: + if limb[1] == 1: # core + continue + body_keypoints_dwpose[limb[1]] = [-1, -1] + subset_dwpose[0][limb[1]] = -1 + else: + if limb[0] == 1: # core + continue + body_keypoints_dwpose[limb[0]] = [-1, -1] + subset_dwpose[0][limb[0]] = -1 + return body_keypoints_dwpose, subset_dwpose + +def correct_full_pose_from_3d(poses_dwpose, poses_3dpose): + ''' + If the bone length of dwpose and 3dpose differ too much, remove the end farthest from the 3d pose. + ''' + poses = [] + for pose_dwpose, pose_3dpose in zip(poses_dwpose, poses_3dpose): + new_candidate, new_subset = correct_body_from_3d(pose_dwpose["bodies"]["candidate"], pose_3dpose["bodies"]["candidate"], pose_dwpose["bodies"]["subset"], pose_3dpose["bodies"]["subset"]) + new_hands_0 = correct_hand_from_3d(pose_dwpose["hands"][0], pose_3dpose["hands"][0]) + new_hands_1 = correct_hand_from_3d(pose_dwpose["hands"][1], pose_3dpose["hands"][1]) + pose = { + "bodies": { + "candidate": new_candidate, + "subset": new_subset + }, + "faces": pose_dwpose["faces"], + "hands": [new_hands_0, new_hands_1] + } + poses.append(pose) + + return poses diff --git a/pose_draw/draw_pose_utils.py b/pose_draw/draw_pose_utils.py new file mode 100644 index 0000000..16b4592 --- /dev/null +++ b/pose_draw/draw_pose_utils.py @@ -0,0 +1,129 @@ +import cv2 +import numpy as np +from PIL import Image +import os +from .draw_utils import draw_bodypose, draw_bodypose_with_feet, draw_handpose_lr, draw_handpose, draw_facepose, draw_bodypose_augmentation + + +def draw_pose(pose, H, W, show_feet=False, show_body=True, show_hand=True, show_face=True, show_cheek=False, dw_bgr=False, dw_hand=False, aug_body_draw=False, optimized_face=False): + final_canvas = np.zeros(shape=(H, W, 3), dtype=np.uint8) + for i in range(len(pose["bodies"]["candidate"])): + canvas = np.zeros(shape=(H, W, 3), dtype=np.uint8) + bodies = pose["bodies"] + faces = pose["faces"][i:i+1] + hands = pose["hands"][2*i:2*i+2] + candidate = bodies["candidate"][i] + subset = bodies["subset"][i:i+1] + + if show_body: + if len(subset[0]) <= 18 or show_feet == False: + if aug_body_draw: + raise NotImplementedError("aug_body_draw is not implemented yet") + else: + canvas = draw_bodypose(canvas, candidate, subset) + else: + canvas = draw_bodypose_with_feet(canvas, candidate, subset) + if dw_bgr: + canvas = cv2.cvtColor(canvas, cv2.COLOR_BGR2RGB) + if show_cheek: + assert show_body == False, "show_cheek and show_body cannot be True at the same time" + canvas = draw_bodypose_augmentation(canvas, candidate, subset, drop_aug=True, shift_aug=False, all_cheek_aug=True) + if show_hand: + if not dw_hand: + canvas = draw_handpose_lr(canvas, hands) + else: + canvas = draw_handpose(canvas, hands) + if show_face: + canvas = draw_facepose(canvas, faces, optimized_face=optimized_face) + final_canvas = final_canvas + canvas + return final_canvas + + +def scale_image_hw_keep_size(img, scale_h, scale_w): + """Scale the image by scale_h and scale_w respectively, keeping the output size unchanged.""" + H, W = img.shape[:2] + new_H, new_W = int(H * scale_h), int(W * scale_w) + scaled = cv2.resize(img, (new_W, new_H), interpolation=cv2.INTER_LINEAR) + + result = np.zeros_like(img) + + # 计算在目标图上的放置范围 + # --- Y方向 --- + if new_H >= H: + y_start_src = (new_H - H) // 2 + y_end_src = y_start_src + H + y_start_dst = 0 + y_end_dst = H + else: + y_start_src = 0 + y_end_src = new_H + y_start_dst = (H - new_H) // 2 + y_end_dst = y_start_dst + new_H + + # --- X方向 --- + if new_W >= W: + x_start_src = (new_W - W) // 2 + x_end_src = x_start_src + W + x_start_dst = 0 + x_end_dst = W + else: + x_start_src = 0 + x_end_src = new_W + x_start_dst = (W - new_W) // 2 + x_end_dst = x_start_dst + new_W + + # 将 scaled 映射到 result + result[y_start_dst:y_end_dst, x_start_dst:x_end_dst] = scaled[y_start_src:y_end_src, x_start_src:x_end_src] + + return result + +def draw_pose_to_canvas_np(poses, pool, H, W, reshape_scale, show_feet_flag=False, show_body_flag=True, show_hand_flag=True, show_face_flag=True, show_cheek_flag=False, dw_bgr=False, dw_hand=False, aug_body_draw=False): + canvas_np_lst = [] + for pose in poses: + if reshape_scale > 0: + pool.apply_random_reshapes(pose) + canvas = draw_pose(pose, H, W, show_feet_flag, show_body_flag, show_hand_flag, show_face_flag, show_cheek_flag, dw_bgr, dw_hand, aug_body_draw, optimized_face=True) + canvas_np_lst.append(canvas) + return canvas_np_lst + + +def draw_pose_to_canvas(poses, pool, H, W, reshape_scale, points_only_flag, show_feet_flag, show_body_flag=True, show_hand_flag=True, show_face_flag=True, show_cheek_flag=False, dw_bgr=False, dw_hand=False, aug_body_draw=False): + canvas_lst = [] + for pose in poses: + if reshape_scale > 0: + pool.apply_random_reshapes(pose) + canvas = draw_pose(pose, H, W, show_feet_flag, show_body_flag, show_hand_flag, show_face_flag, show_cheek_flag, dw_bgr, dw_hand, aug_body_draw, optimized_face=False) + canvas_img = Image.fromarray(canvas) + canvas_lst.append(canvas_img) + return canvas_lst + + +def get_mp4_filenames_from_directory(dwpose_keypoints_dir): + mp4_filenames_dwpose = [] + # Get all available mp4 files by intersecting keypoints and mp4 + if dwpose_keypoints_dir: + for root, dirs, files in os.walk(dwpose_keypoints_dir): + for file in files: + if file.lower().endswith('.pt'): # Only look for .mp4 files + mp4_filenames_dwpose.append(file.replace(".pt", ".mp4")) # Get absolute path + return mp4_filenames_dwpose + +def project_dwpose_to_3d(dwpose_keypoint, original_threed_keypoint, focal, princpt, H, W): + # Camera intrinsic parameters + # fx, fy = focal, focal + fx, fy = focal + cx, cy = princpt + + # 2D keypoint coordinates + x_2d, y_2d = dwpose_keypoint[0] * W, dwpose_keypoint[1] * H + + # Original 3D point (in camera coordinate system) + ori_x, ori_y, ori_z = original_threed_keypoint + + # Use the new 2D point and original depth to compute the new 3D point by back-projection + # Formula: x = (u - cx) * z / fx + new_x = (x_2d - cx) * ori_z / fx + new_y = (y_2d - cy) * ori_z / fy + new_z = ori_z # Keep the depth unchanged + + return [new_x, new_y, new_z] diff --git a/pose_draw/draw_utils.py b/pose_draw/draw_utils.py new file mode 100644 index 0000000..84e6e4f --- /dev/null +++ b/pose_draw/draw_utils.py @@ -0,0 +1,658 @@ +# https://github.com/IDEA-Research/DWPose +import math +import numpy as np +import matplotlib +import cv2 +import random + +eps = 0.01 + + +def smart_resize(x, s): + Ht, Wt = s + if x.ndim == 2: + Ho, Wo = x.shape + Co = 1 + else: + Ho, Wo, Co = x.shape + if Co == 3 or Co == 1: + k = float(Ht + Wt) / float(Ho + Wo) + return cv2.resize( + x, + (int(Wt), int(Ht)), + interpolation=cv2.INTER_AREA if k < 1 else cv2.INTER_LANCZOS4, + ) + else: + return np.stack([smart_resize(x[:, :, i], s) for i in range(Co)], axis=2) + + +def smart_resize_k(x, fx, fy): + if x.ndim == 2: + Ho, Wo = x.shape + Co = 1 + else: + Ho, Wo, Co = x.shape + Ht, Wt = Ho * fy, Wo * fx + if Co == 3 or Co == 1: + k = float(Ht + Wt) / float(Ho + Wo) + return cv2.resize( + x, + (int(Wt), int(Ht)), + interpolation=cv2.INTER_AREA if k < 1 else cv2.INTER_LANCZOS4, + ) + else: + return np.stack([smart_resize_k(x[:, :, i], fx, fy) for i in range(Co)], axis=2) + + +def padRightDownCorner(img, stride, padValue): + h = img.shape[0] + w = img.shape[1] + + pad = 4 * [None] + pad[0] = 0 # up + pad[1] = 0 # left + pad[2] = 0 if (h % stride == 0) else stride - (h % stride) # down + pad[3] = 0 if (w % stride == 0) else stride - (w % stride) # right + + img_padded = img + pad_up = np.tile(img_padded[0:1, :, :] * 0 + padValue, (pad[0], 1, 1)) + img_padded = np.concatenate((pad_up, img_padded), axis=0) + pad_left = np.tile(img_padded[:, 0:1, :] * 0 + padValue, (1, pad[1], 1)) + img_padded = np.concatenate((pad_left, img_padded), axis=1) + pad_down = np.tile(img_padded[-2:-1, :, :] * 0 + padValue, (pad[2], 1, 1)) + img_padded = np.concatenate((img_padded, pad_down), axis=0) + pad_right = np.tile(img_padded[:, -2:-1, :] * 0 + padValue, (1, pad[3], 1)) + img_padded = np.concatenate((img_padded, pad_right), axis=1) + + return img_padded, pad + + +def transfer(model, model_weights): + transfered_model_weights = {} + for weights_name in model.state_dict().keys(): + transfered_model_weights[weights_name] = model_weights[ + ".".join(weights_name.split(".")[1:]) + ] + return transfered_model_weights + +def draw_bodypose_with_feet(canvas, candidate, subset): + H, W, C = canvas.shape + candidate = np.array(candidate) + subset = np.array(subset) + + stickwidth = 4 + + # 原始18个关节点的连接顺序(和 OpenPose 的 COCO 模型一致) + limbSeq = [ + [2, 3], + [2, 6], + [3, 4], + [4, 5], + [6, 7], + [7, 8], + [2, 9], + [9, 10], + [10, 11], + [2, 12], + [12, 13], + [13, 14], + [2, 1], + [1, 15], + [15, 17], + [1, 16], + [16, 18], + [3, 17], + [6, 18], + ] + + # 添加脚部连接线:10->18, 10->19, 10->20;13->21, 13->22, 13->23 + foot_limbSeq = [ + [14, 19], + [14, 20], + [14, 21], + [11, 22], + [11, 23], + [11, 24], + ] + + # 生成颜色(原始18条颜色 + 6条新颜色) + colors = [ + [255, 0, 0], + [255, 85, 0], + [255, 170, 0], + [255, 255, 0], + [170, 255, 0], + [85, 255, 0], + [0, 255, 0], + [0, 255, 85], + [0, 255, 170], + [0, 255, 255], + [0, 170, 255], + [0, 85, 255], + [0, 0, 255], + [85, 0, 255], + [170, 0, 255], + [255, 0, 255], + [255, 0, 170], + [255, 0, 85], + ] + + colors_feet = [ + [100, 0, 215], [80, 0, 235], [60, 0, 255], + [0, 235, 150], [0, 215, 170], [0, 195, 190], + ] + + colors = colors + colors_feet + + for i in range(17): + for n in range(len(subset)): + index = subset[n][np.array(limbSeq[i]) - 1] + if -1 in index: + continue + Y = candidate[index.astype(int), 0] * float(W) + X = candidate[index.astype(int), 1] * float(H) + mX = np.mean(X) + mY = np.mean(Y) + length = ((X[0] - X[1]) ** 2 + (Y[0] - Y[1]) ** 2) ** 0.5 + angle = math.degrees(math.atan2(X[0] - X[1], Y[0] - Y[1])) + polygon = cv2.ellipse2Poly( + (int(mY), int(mX)), (int(length / 2), stickwidth), int(angle), 0, 360, 1 + ) + cv2.fillConvexPoly(canvas, polygon, colors[i]) + + for i in range(6): + for n in range(len(subset)): + index = subset[n][np.array(foot_limbSeq[i]) - 1] + if -1 in index: + continue + Y = candidate[index.astype(int), 0] * float(W) + X = candidate[index.astype(int), 1] * float(H) + mX = np.mean(X) + mY = np.mean(Y) + length = ((X[0] - X[1]) ** 2 + (Y[0] - Y[1]) ** 2) ** 0.5 + angle = math.degrees(math.atan2(X[0] - X[1], Y[0] - Y[1])) + polygon = cv2.ellipse2Poly( + (int(mY), int(mX)), (int(length / 2), stickwidth), int(angle), 0, 360, 1 + ) + cv2.fillConvexPoly(canvas, polygon, colors_feet[i]) + + + canvas = (canvas * 0.6).astype(np.uint8) + + # 画关键点 + for i in range(24): + for n in range(len(subset)): + index = int(subset[n][i]) + if index == -1: + continue + x, y = candidate[index][0:2] + x = int(x * W) + y = int(y * H) + cv2.circle(canvas, (int(x), int(y)), 4, colors[i], thickness=-1) + return canvas + + +def draw_bodypose_augmentation(canvas, candidate, subset, drop_aug=True, shift_aug=False, all_cheek_aug=False): + H, W, C = canvas.shape + candidate = np.array(candidate) + subset = np.array(subset) + + stickwidth = 4 + + limbSeq = [ + [2, 3], # 1->2 左肩 0 + [2, 6], # 1->5 右肩 1 + [3, 4], # 2->3 左臂 2 + [4, 5], # 3->4 左肘 3 + [6, 7], # 5->6 右臂 4 + [7, 8], # 6->7 右肘 5 + [2, 9], # 6 + [9, 10], # 7 + [10, 11], # 8 + [2, 12], # 9 + [12, 13], # 10 + [13, 14], # 11 + [2, 1], # 12 + [1, 15], # 13 cheek + [15, 17], # 14 cheek + [1, 16], # 15 cheek + [16, 18], # 16 cheek + [3, 17], + [6, 18], + ] + + colors = [ + [255, 0, 0], + [255, 85, 0], + [255, 170, 0], + [255, 255, 0], + [170, 255, 0], + [85, 255, 0], + [0, 255, 0], + [0, 255, 85], + [0, 255, 170], + [0, 255, 255], + [0, 170, 255], + [0, 85, 255], + [0, 0, 255], + [85, 0, 255], + [170, 0, 255], + [255, 0, 255], + [255, 0, 170], + [255, 0, 85], + ] + + # 随机选0-2根骨骼进行丢弃 + if drop_aug: + arr_drop = list(range(17)) + k_drop = random.choices([0, 1, 2], weights=[0.5, 0.3, 0.2])[0] + drop_indices = random.sample(arr_drop, k_drop) + else: + drop_indices = [] + if shift_aug: + shift_indices = random.sample(list(range(17)), 2) + else: + shift_indices = [] + if all_cheek_aug: + drop_indices = list(range(13)) # 0-12对应的骨骼都扔掉 + + for i in range(17): + for n in range(len(subset)): + index = subset[n][np.array(limbSeq[i]) - 1] + if -1 in index: + continue + Y = candidate[index.astype(int), 0] * float(W) + X = candidate[index.astype(int), 1] * float(H) + + if i in drop_indices: + continue + + mX = np.mean(X) # 计算两个关节点之间的中点 + mY = np.mean(Y) + length = ((X[0] - X[1]) ** 2 + (Y[0] - Y[1]) ** 2) ** 0.5 + if i in shift_indices: + mX = mX + random.uniform(-length/4, length/4) + mY = mY + random.uniform(-length/4, length/4) + angle = math.degrees(math.atan2(X[0] - X[1], Y[0] - Y[1])) + polygon = cv2.ellipse2Poly( + (int(mY), int(mX)), (int(length / 2), stickwidth), int(angle), 0, 360, 1 + ) + cv2.fillConvexPoly(canvas, polygon, colors[i]) + + canvas = (canvas * 0.6).astype(np.uint8) + + for i in range(18): + if all_cheek_aug: + if not i in [0, 14, 15, 16, 17]: + continue + for n in range(len(subset)): + index = int(subset[n][i]) + if index == -1: + continue + x, y = candidate[index][0:2] + x = int(x * W) + y = int(y * H) + cv2.circle(canvas, (int(x), int(y)), 4, colors[i], thickness=-1) + + return canvas + +def draw_bodypose(canvas, candidate, subset): + H, W, C = canvas.shape + candidate = np.array(candidate) + subset = np.array(subset) + + stickwidth = 4 + + limbSeq = [ + [2, 3], + [2, 6], + [3, 4], + [4, 5], + [6, 7], + [7, 8], + [2, 9], + [9, 10], + [10, 11], + [2, 12], + [12, 13], + [13, 14], + [2, 1], + [1, 15], + [15, 17], + [1, 16], + [16, 18], + [3, 17], + [6, 18], + ] + + colors = [ + [255, 0, 0], + [255, 85, 0], + [255, 170, 0], + [255, 255, 0], + [170, 255, 0], + [85, 255, 0], + [0, 255, 0], + [0, 255, 85], + [0, 255, 170], + [0, 255, 255], + [0, 170, 255], + [0, 85, 255], + [0, 0, 255], + [85, 0, 255], + [170, 0, 255], + [255, 0, 255], + [255, 0, 170], + [255, 0, 85], + ] + + for i in range(17): + for n in range(len(subset)): + index = subset[n][np.array(limbSeq[i]) - 1] + if -1 in index: + continue + Y = candidate[index.astype(int), 0] * float(W) + X = candidate[index.astype(int), 1] * float(H) + mX = np.mean(X) + mY = np.mean(Y) + length = ((X[0] - X[1]) ** 2 + (Y[0] - Y[1]) ** 2) ** 0.5 + angle = math.degrees(math.atan2(X[0] - X[1], Y[0] - Y[1])) + polygon = cv2.ellipse2Poly( + (int(mY), int(mX)), (int(length / 2), stickwidth), int(angle), 0, 360, 1 + ) + cv2.fillConvexPoly(canvas, polygon, colors[i]) + + canvas = (canvas * 0.6).astype(np.uint8) + + for i in range(18): + for n in range(len(subset)): + index = int(subset[n][i]) + if index == -1: + continue + x, y = candidate[index][0:2] + x = int(x * W) + y = int(y * H) + cv2.circle(canvas, (int(x), int(y)), 4, colors[i], thickness=-1) + + return canvas + +def draw_handpose_lr(canvas, all_hand_peaks): + H, W, C = canvas.shape + + # 连接顺序:21个关键点的骨架连线 + edges = [ + [0, 1], [1, 2], [2, 3], [3, 4], + [0, 5], [5, 6], [6, 7], [7, 8], + [0, 9], [9, 10], [10, 11], [11, 12], + [0, 13], [13, 14], [14, 15], [15, 16], + [0, 17], [17, 18], [18, 19], [19, 20], + ] + + all_num_hands = len(all_hand_peaks) + for peaks_idx, peaks in enumerate(all_hand_peaks): + left_or_right = not (peaks_idx >= all_num_hands / 2) + base_hue = 0 if left_or_right == 0 else 0.3 + peaks = np.array(peaks) + + for ie, e in enumerate(edges): + x1, y1 = peaks[e[0]] + x2, y2 = peaks[e[1]] + x1 = int(x1 * W) + y1 = int(y1 * H) + x2 = int(x2 * W) + y2 = int(y2 * H) + if x1 > eps and y1 > eps and x2 > eps and y2 > eps: + if left_or_right == 0: + hsv_color = [ (base_hue + ie / float(len(edges)) * 0.8), 0.9, 0.9 ] + else: + hsv_color = [ (base_hue + ie / float(len(edges)) * 0.8), 0.8, 1 ] + rgb_color = matplotlib.colors.hsv_to_rgb(hsv_color) * 255 + cv2.line( + canvas, + (x1, y1), + (x2, y2), + rgb_color, + thickness=2, + ) + + for i, keypoint in enumerate(peaks): + x, y = keypoint + x = int(x * W) + y = int(y * H) + if x > eps and y > eps: + # 关键点也用淡色标注(左手蓝、右手红) + point_color = (245, 100, 100) if left_or_right == 0 else (100, 100, 255) + cv2.circle(canvas, (x, y), 4, point_color, thickness=-1) + + return canvas + +def draw_handpose(canvas, all_hand_peaks): + H, W, C = canvas.shape + stickwidth_thin = min(max(int(min(H, W) / 300), 1), 2) + + edges = [ + [0, 1], + [1, 2], + [2, 3], + [3, 4], + [0, 5], + [5, 6], + [6, 7], + [7, 8], + [0, 9], + [9, 10], + [10, 11], + [11, 12], + [0, 13], + [13, 14], + [14, 15], + [15, 16], + [0, 17], + [17, 18], + [18, 19], + [19, 20], + ] + + for peaks in all_hand_peaks: + peaks = np.array(peaks) + + for ie, e in enumerate(edges): + x1, y1 = peaks[e[0]] + x2, y2 = peaks[e[1]] + x1 = int(x1 * W) + y1 = int(y1 * H) + x2 = int(x2 * W) + y2 = int(y2 * H) + if x1 > eps and y1 > eps and x2 > eps and y2 > eps: + cv2.line( + canvas, + (x1, y1), + (x2, y2), + matplotlib.colors.hsv_to_rgb([ie / float(len(edges)), 1.0, 1.0]) + * 255, + thickness=stickwidth_thin, + ) + + for i, keyponit in enumerate(peaks): + x, y = keyponit + x = int(x * W) + y = int(y * H) + if x > eps and y > eps: + cv2.circle(canvas, (x, y), stickwidth_thin, (0, 0, 255), thickness=-1) + return canvas + + +def draw_facepose(canvas, all_lmks, optimized_face=True): + H, W, C = canvas.shape + stickwidth = min(max(int(min(H, W) / 200), 1), 3) + stickwidth_thin = min(max(int(min(H, W) / 300), 1), 2) + + for lmks in all_lmks: + lmks = np.array(lmks) + for lmk_idx, lmk in enumerate(lmks): + x, y = lmk + x = int(x * W) + y = int(y * H) + if x > eps and y > eps: + if optimized_face: + if lmk_idx in list(range(17, 27)) + list(range(36, 70)): + cv2.circle(canvas, (x, y), stickwidth_thin, (255, 255, 255), thickness=-1) + else: + cv2.circle(canvas, (x, y), stickwidth, (255, 255, 255), thickness=-1) + return canvas + + + + + +# detect hand according to body pose keypoints +# please refer to https://github.com/CMU-Perceptual-Computing-Lab/openpose/blob/master/src/openpose/hand/handDetector.cpp +def handDetect(candidate, subset, oriImg): + # right hand: wrist 4, elbow 3, shoulder 2 + # left hand: wrist 7, elbow 6, shoulder 5 + ratioWristElbow = 0.33 + detect_result = [] + image_height, image_width = oriImg.shape[0:2] + for person in subset.astype(int): + # if any of three not detected + has_left = np.sum(person[[5, 6, 7]] == -1) == 0 + has_right = np.sum(person[[2, 3, 4]] == -1) == 0 + if not (has_left or has_right): + continue + hands = [] + # left hand + if has_left: + left_shoulder_index, left_elbow_index, left_wrist_index = person[[5, 6, 7]] + x1, y1 = candidate[left_shoulder_index][:2] + x2, y2 = candidate[left_elbow_index][:2] + x3, y3 = candidate[left_wrist_index][:2] + hands.append([x1, y1, x2, y2, x3, y3, True]) + # right hand + if has_right: + right_shoulder_index, right_elbow_index, right_wrist_index = person[ + [2, 3, 4] + ] + x1, y1 = candidate[right_shoulder_index][:2] + x2, y2 = candidate[right_elbow_index][:2] + x3, y3 = candidate[right_wrist_index][:2] + hands.append([x1, y1, x2, y2, x3, y3, False]) + + for x1, y1, x2, y2, x3, y3, is_left in hands: + # pos_hand = pos_wrist + ratio * (pos_wrist - pos_elbox) = (1 + ratio) * pos_wrist - ratio * pos_elbox + # handRectangle.x = posePtr[wrist*3] + ratioWristElbow * (posePtr[wrist*3] - posePtr[elbow*3]); + # handRectangle.y = posePtr[wrist*3+1] + ratioWristElbow * (posePtr[wrist*3+1] - posePtr[elbow*3+1]); + # const auto distanceWristElbow = getDistance(poseKeypoints, person, wrist, elbow); + # const auto distanceElbowShoulder = getDistance(poseKeypoints, person, elbow, shoulder); + # handRectangle.width = 1.5f * fastMax(distanceWristElbow, 0.9f * distanceElbowShoulder); + x = x3 + ratioWristElbow * (x3 - x2) + y = y3 + ratioWristElbow * (y3 - y2) + distanceWristElbow = math.sqrt((x3 - x2) ** 2 + (y3 - y2) ** 2) + distanceElbowShoulder = math.sqrt((x2 - x1) ** 2 + (y2 - y1) ** 2) + width = 1.5 * max(distanceWristElbow, 0.9 * distanceElbowShoulder) + # x-y refers to the center --> offset to topLeft point + # handRectangle.x -= handRectangle.width / 2.f; + # handRectangle.y -= handRectangle.height / 2.f; + x -= width / 2 + y -= width / 2 # width = height + # overflow the image + if x < 0: + x = 0 + if y < 0: + y = 0 + width1 = width + width2 = width + if x + width > image_width: + width1 = image_width - x + if y + width > image_height: + width2 = image_height - y + width = min(width1, width2) + # the max hand box value is 20 pixels + if width >= 20: + detect_result.append([int(x), int(y), int(width), is_left]) + + """ + return value: [[x, y, w, True if left hand else False]]. + width=height since the network require squared input. + x, y is the coordinate of top left + """ + return detect_result + + +# Written by Lvmin +def faceDetect(candidate, subset, oriImg): + # left right eye ear 14 15 16 17 + detect_result = [] + image_height, image_width = oriImg.shape[0:2] + for person in subset.astype(int): + has_head = person[0] > -1 + if not has_head: + continue + + has_left_eye = person[14] > -1 + has_right_eye = person[15] > -1 + has_left_ear = person[16] > -1 + has_right_ear = person[17] > -1 + + if not (has_left_eye or has_right_eye or has_left_ear or has_right_ear): + continue + + head, left_eye, right_eye, left_ear, right_ear = person[[0, 14, 15, 16, 17]] + + width = 0.0 + x0, y0 = candidate[head][:2] + + if has_left_eye: + x1, y1 = candidate[left_eye][:2] + d = max(abs(x0 - x1), abs(y0 - y1)) + width = max(width, d * 3.0) + + if has_right_eye: + x1, y1 = candidate[right_eye][:2] + d = max(abs(x0 - x1), abs(y0 - y1)) + width = max(width, d * 3.0) + + if has_left_ear: + x1, y1 = candidate[left_ear][:2] + d = max(abs(x0 - x1), abs(y0 - y1)) + width = max(width, d * 1.5) + + if has_right_ear: + x1, y1 = candidate[right_ear][:2] + d = max(abs(x0 - x1), abs(y0 - y1)) + width = max(width, d * 1.5) + + x, y = x0, y0 + + x -= width + y -= width + + if x < 0: + x = 0 + + if y < 0: + y = 0 + + width1 = width * 2 + width2 = width * 2 + + if x + width > image_width: + width1 = image_width - x + + if y + width > image_height: + width2 = image_height - y + + width = min(width1, width2) + + if width >= 20: + detect_result.append([int(x), int(y), int(width)]) + + return detect_result + + +# get max index of 2d array +def npmax(array): + arrayindex = array.argmax(1) + arrayvalue = array.max(1) + i = arrayvalue.argmax() + j = arrayindex[i] + return i, j diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..1bee94c --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,15 @@ +[project] +name = "ComfyUI-SCAIL-Pose" +description = "ComfyUI nodes for SCAIL input processing" +version = "1.0.0" +license = {file = "LICENSE"} +dependencies = ["taichi", "pyrender", "trimesh", "opencv-python", "matplotlib", "pillow"] + +[project.urls] +Repository = "https://github.com/kijai/ComfyUI-SCAIL-Pose" +# Used by Comfy Registry https://comfyregistry.org + +[tool.comfy] +PublisherId = "kijai" +DisplayName = "ComfyUI-SCAIL-Pose" +Icon = "" diff --git a/readme.md b/readme.md new file mode 100644 index 0000000..d5cf92e --- /dev/null +++ b/readme.md @@ -0,0 +1,12 @@ +# ComfyUI nodes for SCAIL-pose processing + + +The code is cleaned, simplified version of: https://github.com/zai-org/SCAIL-Pose + +For face and hands, instead of DWPose this uses Vitpose and it's outputs converted into DWpose format for the optional alignment + +VitPose detector is available in these nodes: https://github.com/kijai/ComfyUI-WanAnimatePreprocess + +NLF model loader is already included in WanVideoWrapper + +Reason this is separate repository is the additional requirements of `taichi` and `pyrender` diff --git a/render_3d/render_cylinder.py b/render_3d/render_cylinder.py new file mode 100644 index 0000000..65a9964 --- /dev/null +++ b/render_3d/render_cylinder.py @@ -0,0 +1,93 @@ +import numpy as np +import cv2 +from PIL import Image +import pyrender +import trimesh + +def render_colored_cylinders(cylinder_specs, focal, princpt, image_size=(1280, 1280), img=None): + + H, W = image_size + if isinstance(focal, float) or isinstance(focal, int): + fx, fy = focal, focal + else: + fx, fy = focal[0], focal[1] + cx, cy = princpt + + + # Initialize scene + scene = pyrender.Scene(bg_color=[0, 0, 0, 0], ambient_light=[0.1, 0.1, 0.1]) + + + # Set up camera + camera = pyrender.IntrinsicsCamera(fx=fx, fy=fy, cx=cx, cy=cy, znear=0.5, zfar=10000) + pyrender2opencv = np.array([[1.0, 0, 0, 0], + [0, -1, 0, 0], + [0, 0, -1, 0], + [0, 0, 0, 1]]) + cam_pose = pyrender2opencv @ np.eye(4) + scene.add(camera, pose=cam_pose) + + # Add light source + light = pyrender.DirectionalLight(color=np.ones(3), intensity=3.0) + scene.add(light, pose=cam_pose) + + points_to_draw = [] + + for start, end, color in cylinder_specs: + start = np.array(start) + end = np.array(end) + vec = end - start + height = np.linalg.norm(vec) + if height == 0: + continue + + tm = trimesh.creation.cylinder(radius=12, height=height, sections=16) + + # Rotate to align with z-axis + z_axis = np.array([0, 0, 1]) + axis = np.cross(z_axis, vec) + if np.linalg.norm(axis) > 1e-6: + axis = axis / np.linalg.norm(axis) + angle = np.arccos(np.dot(z_axis, vec) / height) + rot = trimesh.transformations.rotation_matrix(angle, axis) + tm.apply_transform(rot) + + tm.apply_translation(start + vec / 2) + + # Material color (supports RGBA) + rgba = np.array(color) + material = pyrender.MetallicRoughnessMaterial( + metallicFactor=0.1, + roughnessFactor=0.5, + baseColorFactor=rgba + ) + + mesh = pyrender.Mesh.from_trimesh(tm, material=material) + scene.add(mesh) + + # Projected points for visualization, check if projection is correct + x1 = fx * (start[0] / start[2]) + cx + y1 = fy * (start[1] / start[2]) + cy + x2 = fx * (end[0] / end[2]) + cx + y2 = fy * (end[1] / end[2]) + cy + points_to_draw.append((x1, y1)) + points_to_draw.append((x2, y2)) + + + # Render + r = pyrender.OffscreenRenderer(viewport_width=W, viewport_height=H, point_size=1.0) + color, _ = r.render(scene, flags=pyrender.RenderFlags.RGBA) + + # Post-processing + color = color.astype(np.float32) / 255.0 + final_img = (color * 255).astype(np.uint8) + + # Draw points, check if projection is correct + for (x, y) in points_to_draw: + print(f" debug point: {x}, {y}") + x_draw = int(x) + y_draw = int(y) + cv2.circle(final_img, (x_draw, y_draw), radius=4, color=(0, 255, 0), thickness=-1) + + return Image.fromarray(final_img) + diff --git a/render_3d/taichi_cylinder.py b/render_3d/taichi_cylinder.py new file mode 100644 index 0000000..1c99210 --- /dev/null +++ b/render_3d/taichi_cylinder.py @@ -0,0 +1,207 @@ +import taichi as ti +import numpy as np +import random +import math + +ti.init(arch=ti.cuda) + +def flatten_specs(specs_list): + """把 specs_list 拉平为 numpy 数组 + 索引表""" + starts, ends, colors = [], [], [] + frame_offset, frame_count = [], [] + offset = 0 + for specs in specs_list: + frame_offset.append(offset) + frame_count.append(len(specs)) + for (s, e, c) in specs: + starts.append(s) + ends.append(e) + colors.append(c) + offset += len(specs) + return ( + np.array(starts, dtype=np.float32), + np.array(ends, dtype=np.float32), + np.array(colors, dtype=np.float32), + np.array(frame_offset, dtype=np.int32), + np.array(frame_count, dtype=np.int32), + ) + +def render_whole(specs_list, H=480, W=640, fx=500, fy=500, cx=240, cy=320, radius=21.5): + img = ti.Vector.field(4, dtype=ti.f32, shape=(H, W)) + starts, ends, colors, frame_offset, frame_count = flatten_specs(specs_list) + total_cyl = len(starts) + n_frames = len(specs_list) + z_min = min(starts[:, 2].min(), ends[:, 2].min()) + z_max = max(starts[:, 2].max(), ends[:, 2].max()) + + # ========= 相机内参 ========= + znear = 0.1 + zfar = max(min(z_max, 25000), 10000) + C = ti.Vector([0.0, 0.0, 0.0]) # 相机中心 + light_dir = ti.Vector([0.0, 0.0, 1.0]) + + c_start = ti.Vector.field(3, dtype=ti.f32, shape=total_cyl) + c_end = ti.Vector.field(3, dtype=ti.f32, shape=total_cyl) + c_rgba = ti.Vector.field(4, dtype=ti.f32, shape=total_cyl) + n_cyl = ti.field(dtype=ti.i32, shape=()) # 实际数量 + f_offset = ti.field(dtype=ti.i32, shape=n_frames) + f_count = ti.field(dtype=ti.i32, shape=n_frames) + frame_id = ti.field(dtype=ti.i32, shape=()) # 当前帧号 + z_min_field = ti.field(dtype=ti.f32, shape=()) + z_max_field = ti.field(dtype=ti.f32, shape=()) + + z_min_field[None] = z_min + z_max_field[None] = z_max + + # # ====== 拷贝数据一次 ====== + c_start.from_numpy(starts) + c_end.from_numpy(ends) + c_rgba.from_numpy(colors) + f_offset.from_numpy(frame_offset) + f_count.from_numpy(frame_count) + + @ti.func + def sd_cylinder(p, a, b, r): + pa = p - a + ba = b - a + h = ba.norm() + eps = 1e-8 + res = 0.0 + if h < eps: + res = pa.norm() - r + else: + ba_n = ba / h + proj = pa.dot(ba_n) + proj_clamped = min(max(proj, 0.0), h) + res = (pa - proj_clamped * ba_n).norm() - r + return res + + @ti.func + def scene_sdf(p): + best_d = 1e6 + best_col = ti.Vector([0.0, 0.0, 0.0, 0.0]) + fid = frame_id[None] # 从 field 里读出来,变成一个普通 int + off = f_offset[fid] + cnt = f_count[fid] + for i in range(cnt): # 只遍历实际数量 + a = c_start[off + i] + b = c_end[off + i] + r = radius + col = c_rgba[off + i] + d = sd_cylinder(p, a, b, r) + if d < best_d: + best_d = d + best_col = col + return best_d, best_col + + @ti.func + def get_normal(p): + e = 1e-3 + dx = scene_sdf(p + ti.Vector([e, 0.0, 0.0]))[0] - scene_sdf(p - ti.Vector([e, 0.0, 0.0]))[0] + dy = scene_sdf(p + ti.Vector([0.0, e, 0.0]))[0] - scene_sdf(p - ti.Vector([0.0, e, 0.0]))[0] + dz = scene_sdf(p + ti.Vector([0.0, 0.0, e]))[0] - scene_sdf(p - ti.Vector([0.0, 0.0, e]))[0] + n = ti.Vector([dx, dy, dz]) + return n.normalized() + + @ti.func + def pixel_to_ray(xi, yi): + u = (xi - cx) / fx + v = (yi - cy) / fy + dir_cam = ti.Vector([u, v, 1.0]).normalized() + Rcw = ti.Matrix.identity(ti.f32, 3) + rd_world = Rcw @ dir_cam + ro_world = C + return ro_world, rd_world + + @ti.kernel + def render(): + depth_near, depth_far = ti.max(z_min_field[None], 0.1), ti.min(z_max_field[None] + 6000, 20000) # 能渲染出来的点,最大12000 + for y, x in img: + ro, rd = pixel_to_ray(x, y) + t = znear + col_out = ti.Vector([0.0, 0.0, 0.0, 0.0]) + for _ in range(300): + p = ro + rd * t + d, col = scene_sdf(p) + if d < 1e-3: + # n = get_normal(p) + # diff = max(n.dot(-light_dir), 0.0) + # lit = 0.3 + 0.7 * diff + # col_out = ti.Vector([col.x * lit, col.y * lit, col.z * lit, col.w]) + # break + + n = get_normal(p) + diff = max(n.dot(-light_dir), 0.0) + + # === Blinn-Phong 镜面反射 === + view_dir = -rd.normalized() + half_dir = (view_dir + -light_dir).normalized() + spec = max(n.dot(half_dir), 0.0) ** 32 # shininess=32,越小越散,越大越锐 + + depth_factor = 1.0 - (p.z - depth_near) / (depth_far - znear) + depth_factor = ti.max(0.0, ti.min(1.0, depth_factor)) + + # 原来的 diffuse/ambient 光照 + diffuse_term = 0.3 + 0.7 * diff + base = col.xyz * diffuse_term * depth_factor + + # 镜面高光(叠加到原有结果上) + highlight = ti.Vector([1.0, 1.0, 1.0]) * (0.5 * spec) * depth_factor + + col_out = ti.Vector([base.x + highlight.x, + base.y + highlight.y, + base.z + highlight.z, + col.w]) + break + + if t > zfar: + break + t += max(d, 1e-4) + img[y, x] = col_out + + frames_np_rgba = [] + for f in range(len(specs_list)): + # start_time = time.time() + frame_id[None] = f + render() + arr = np.clip(img.to_numpy(), 0, 1) + # end_time = time.time() + # print(f"Frame {f} time: {end_time - start_time} seconds") + arr8 = (arr * 255).astype(np.uint8) + frames_np_rgba.append(arr8) + + return frames_np_rgba + + +def random_cylinder(): + """生成一根随机圆柱 (start, end, color)。""" + # 起点 [-200,200]^2, z 在 [-300,-100] + ax = random.uniform(-200, 200) + ay = random.uniform(-200, 200) + az = random.uniform(300, 400) + start = [ax, ay, az] + + # 随机方向和长度 + theta = random.uniform(0, 2*math.pi) + phi = random.uniform(-math.pi/4, math.pi/4) # 倾斜角 + L = 100 + dx = math.cos(phi) * math.cos(theta) + dy = math.cos(phi) * math.sin(theta) + dz = math.sin(phi) + end = [ax + dx * L, ay + dy * L, az + dz * L] + + # 随机颜色 (RGB + alpha=1) + color = [random.random(), random.random(), random.random(), 1.0] + + return (start, end, color) + +def generate_specs_list(num_frames=120, min_cyl=10, max_cyl=120): + """生成 specs_list,每帧有若干随机圆柱.""" + specs_list = [] + for _ in range(num_frames): + n_cyl = random.randint(min_cyl, max_cyl) + specs = [random_cylinder() for _ in range(n_cyl)] + specs_x_shift = [([spec[0][0] + 50, spec[0][1], spec[0][2]], [spec[1][0] + 50, spec[1][1], spec[1][2]], spec[2]) for spec in specs] + specs_list.append(specs) + specs_list.append(specs_x_shift) + return specs_list diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..7f0ba37 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,6 @@ +taichi +pyrender +trimesh +opencv-python +matplotlib +pillow \ No newline at end of file diff --git a/vitpose_utils/utils.py b/vitpose_utils/utils.py new file mode 100644 index 0000000..1a88b4e --- /dev/null +++ b/vitpose_utils/utils.py @@ -0,0 +1,191 @@ + + +import numpy as np +import cv2 + +# source +# https://github.com/Wan-Video/Wan2.2/blob/e9783574ef77be11fcab9aa5607905402538c08d/wan/modules/animate/preprocess/pose2d_utils.py#L1034 + +def bbox_from_detector(bbox, input_resolution=(224, 224), rescale=1.25): + """ + Get center and scale of bounding box from bounding box. + The expected format is [min_x, min_y, max_x, max_y]. + """ + CROP_IMG_HEIGHT, CROP_IMG_WIDTH = input_resolution + CROP_ASPECT_RATIO = CROP_IMG_HEIGHT / float(CROP_IMG_WIDTH) + + # center + center_x = (bbox[0] + bbox[2]) / 2.0 + center_y = (bbox[1] + bbox[3]) / 2.0 + center = np.array([center_x, center_y]) + + # scale + bbox_w = bbox[2] - bbox[0] + bbox_h = bbox[3] - bbox[1] + bbox_size = max(bbox_w * CROP_ASPECT_RATIO, bbox_h) + + scale = np.array([bbox_size / CROP_ASPECT_RATIO, bbox_size]) / 200.0 + # scale = bbox_size / 200.0 + # adjust bounding box tightness + scale *= rescale + return center, scale + +def get_transform(center, scale, res, rot=0): + """Generate transformation matrix.""" + # res: (height, width), (rows, cols) + crop_aspect_ratio = res[0] / float(res[1]) + h = 200 * scale + w = h / crop_aspect_ratio + t = np.zeros((3, 3)) + t[0, 0] = float(res[1]) / w + t[1, 1] = float(res[0]) / h + t[0, 2] = res[1] * (-float(center[0]) / w + .5) + t[1, 2] = res[0] * (-float(center[1]) / h + .5) + t[2, 2] = 1 + if not rot == 0: + rot = -rot # To match direction of rotation from cropping + rot_mat = np.zeros((3, 3)) + rot_rad = rot * np.pi / 180 + sn, cs = np.sin(rot_rad), np.cos(rot_rad) + rot_mat[0, :2] = [cs, -sn] + rot_mat[1, :2] = [sn, cs] + rot_mat[2, 2] = 1 + # Need to rotate around center + t_mat = np.eye(3) + t_mat[0, 2] = -res[1] / 2 + t_mat[1, 2] = -res[0] / 2 + t_inv = t_mat.copy() + t_inv[:2, 2] *= -1 + t = np.dot(t_inv, np.dot(rot_mat, np.dot(t_mat, t))) + return t + +def transform(pt, center, scale, res, invert=0, rot=0): + """Transform pixel location to different reference.""" + t = get_transform(center, scale, res, rot=rot) + if invert: + t = np.linalg.inv(t) + new_pt = np.array([pt[0] - 1, pt[1] - 1, 1.]).T + new_pt = np.dot(t, new_pt) + return np.array([round(new_pt[0]), round(new_pt[1])], dtype=int) + 1 + +def crop(img, center, scale, res): + """ + Crop image according to the supplied bounding box. + res: [rows, cols] + """ + # Upper left point + ul = np.array(transform([1, 1], center, max(scale), res, invert=1)) - 1 + # Bottom right point + br = np.array(transform([res[1] + 1, res[0] + 1], center, max(scale), res, invert=1)) - 1 + + new_shape = [br[1] - ul[1], br[0] - ul[0]] + if len(img.shape) > 2: + new_shape += [img.shape[2]] + new_img = np.zeros(new_shape, dtype=np.float32) + + # Range to fill new array + new_x = max(0, -ul[0]), min(br[0], len(img[0])) - ul[0] + new_y = max(0, -ul[1]), min(br[1], len(img)) - ul[1] + # Range to sample from original image + old_x = max(0, ul[0]), min(len(img[0]), br[0]) + old_y = max(0, ul[1]), min(len(img), br[1]) + try: + new_img[new_y[0]:new_y[1], new_x[0]:new_x[1]] = img[old_y[0]:old_y[1], old_x[0]:old_x[1]] + except Exception as e: + print(e) + + new_img = cv2.resize(new_img, (res[1], res[0])) # (cols, rows) + return new_img, new_shape, (old_x, old_y), (new_x, new_y) # , ul, br + + +def split_kp2ds_for_aa(kp2ds, ret_face=False): + kp2ds_body = (kp2ds[[0, 6, 6, 8, 10, 5, 7, 9, 12, 14, 16, 11, 13, 15, 2, 1, 4, 3, 17, 20]] + kp2ds[[0, 5, 6, 8, 10, 5, 7, 9, 12, 14, 16, 11, 13, 15, 2, 1, 4, 3, 18, 21]]) / 2 + kp2ds_lhand = kp2ds[91:112] + kp2ds_rhand = kp2ds[112:133] + kp2ds_face = kp2ds[22:91] + if ret_face: + return kp2ds_body.copy(), kp2ds_lhand.copy(), kp2ds_rhand.copy(), kp2ds_face.copy() + return kp2ds_body.copy(), kp2ds_lhand.copy(), kp2ds_rhand.copy() + + +def load_pose_metas_from_kp2ds_seq(kp2ds_seq, width, height): + metas = [] + last_kp2ds_body = None + for kps in kp2ds_seq: + kps = kps.copy() + kps[:, 0] /= width + kps[:, 1] /= height + kp2ds_body, kp2ds_lhand, kp2ds_rhand, kp2ds_face = split_kp2ds_for_aa(kps, ret_face=True) + + # Exclude cases where all values are less than 0 + if last_kp2ds_body is not None and kp2ds_body[:, :2].min(axis=1).max() < 0: + kp2ds_body = last_kp2ds_body + last_kp2ds_body = kp2ds_body + + meta = { + "width": width, + "height": height, + "keypoints_body": kp2ds_body, + "keypoints_left_hand": kp2ds_lhand, + "keypoints_right_hand": kp2ds_rhand, + "keypoints_face": kp2ds_face, + } + metas.append(meta) + return metas + +def aaposemeta_to_dwpose_scail(meta): + """ + Convert AA pose metadata to DWpose format matching DWposeDetector output. + + DWpose format: + - bodies: dict with 'candidate' (n, 24, 2) and 'subset' (n, 24) where subset contains indices + - hands: array (2*n, 21, 2) - stacked right/left hands + - faces: array (n, 68, 2) + """ + # Body keypoints (excluding last 2) + candidate_body = meta['keypoints_body'][:-2][:, :2] # (24, 2) + score_body = meta['keypoints_body'][:-2][:, 2] # (24,) + + # Create subset: contains joint index if visible, -1 if not + subset_body = np.arange(len(candidate_body), dtype=float) + subset_body[score_body <= 0.3] = -1 # Match DWpose threshold + + # Bodies dict with single person (expand to match multi-person format) + bodies = { + "candidate": np.expand_dims(candidate_body, axis=0), # (1, 24, 2) + "subset": np.expand_dims(subset_body, axis=0) # (1, 24) + } + + # Hands: stack right then left (2, 21, 2) + hands_coords = np.stack([ + meta['keypoints_right_hand'][:, :2], + meta['keypoints_left_hand'][:, :2] + ], axis=0) + + hands_score = np.stack([ + meta['keypoints_right_hand'][:, 2], + meta['keypoints_left_hand'][:, 2] + ], axis=0) + + # Faces: (1, 68, 2) - skip first face keypoint like DWpose does (24:92 = 68 points) + faces_coords = np.expand_dims(meta['keypoints_face'][1:][:, :2], axis=0) + faces_score = np.expand_dims(meta['keypoints_face'][1:][:, 2], axis=0) + + # Match DWpose output structure + dwpose_format = { + "bodies": bodies, + "hands": hands_coords, + "faces": faces_coords + } + + # Optional: include scores separately like DWpose does + score_dict = { + "body_score": np.expand_dims(score_body, axis=0), + "hand_score": hands_score, + "face_score": faces_score + } + + # Merge score dict into dwpose_format + dwpose_format.update(score_dict) + + return dwpose_format